{
  "name": "AI Prep Buddy Dataset",
  "version": "2.0.0",
  "total_questions": 1801,
  "total_sections": 58,
  "questions": [
    {
      "id": 1,
      "section": "Section 1 — Strategy, Vision & Technical Leadership",
      "question": "How would you build a 2–3 year AI roadmap for this org, and how do you sequence build-vs-buy decisions?",
      "difficulty": 2
    },
    {
      "id": 2,
      "section": "Section 1 — Strategy, Vision & Technical Leadership",
      "question": "How do you decide when a problem needs fine-tuning vs. RAG vs. prompt engineering vs. classic ML vs. no ML?",
      "difficulty": 2
    },
    {
      "id": 3,
      "section": "Section 1 — Strategy, Vision & Technical Leadership",
      "question": "Walk through how you'd evaluate and select a foundation-model provider (cost, latency, quality, data residency, lock-in).",
      "difficulty": 3
    },
    {
      "id": 4,
      "section": "Section 1 — Strategy, Vision & Technical Leadership",
      "question": "How do you set and defend an AI platform's technical principles (model-agnostic layer, single vector-store standard, etc.)?",
      "difficulty": 2
    },
    {
      "id": 5,
      "section": "Section 1 — Strategy, Vision & Technical Leadership",
      "question": "How do you communicate AI capability limits to executives who overestimate what AI can do?",
      "difficulty": 2
    },
    {
      "id": 6,
      "section": "Section 1 — Strategy, Vision & Technical Leadership",
      "question": "Center of excellence vs. embedded AI engineers across product teams — tradeoffs?",
      "difficulty": 2
    },
    {
      "id": 7,
      "section": "Section 1 — Strategy, Vision & Technical Leadership",
      "question": "How do you evaluate ROI on a proposed GenAI initiative before committing headcount?",
      "difficulty": 2
    },
    {
      "id": 8,
      "section": "Section 1 — Strategy, Vision & Technical Leadership",
      "question": "How would you architect for multi-cloud/model-provider portability without exploding cost or complexity?",
      "difficulty": 2
    },
    {
      "id": 9,
      "section": "Section 1 — Strategy, Vision & Technical Leadership",
      "question": "How do you decide what NOT to build in-house on an AI platform team?",
      "difficulty": 3
    },
    {
      "id": 10,
      "section": "Section 1 — Strategy, Vision & Technical Leadership",
      "question": "What's your framework for prioritizing a backlog of 20 competing AI use cases?",
      "difficulty": 2
    },
    {
      "id": 11,
      "section": "Section 1 — Strategy, Vision & Technical Leadership",
      "question": "How do you structure a build vs. partner vs. acquire decision for a critical AI capability?",
      "difficulty": 2
    },
    {
      "id": 12,
      "section": "Section 1 — Strategy, Vision & Technical Leadership",
      "question": "How do you set technical OKRs for an AI platform team that are outcome-based, not activity-based?",
      "difficulty": 2
    },
    {
      "id": 13,
      "section": "Section 1 — Strategy, Vision & Technical Leadership",
      "question": "How would you design an internal AI platform that serves both data scientists and product engineers?",
      "difficulty": 2
    },
    {
      "id": 14,
      "section": "Section 1 — Strategy, Vision & Technical Leadership",
      "question": "How do you decide the org's stance on open-weight vs. closed frontier models?",
      "difficulty": 2
    },
    {
      "id": 15,
      "section": "Section 1 — Strategy, Vision & Technical Leadership",
      "question": "What's your approach to sunset-ing a legacy ML system in favor of a GenAI-based one?",
      "difficulty": 2
    },
    {
      "id": 16,
      "section": "Section 1 — Strategy, Vision & Technical Leadership",
      "question": "How do you build a business case for investing in evaluation infrastructure before launch pressure hits?",
      "difficulty": 2
    },
    {
      "id": 17,
      "section": "Section 1 — Strategy, Vision & Technical Leadership",
      "question": "How do you decide between a monolithic \"AI platform\" and a set of loosely coupled AI services?",
      "difficulty": 3
    },
    {
      "id": 18,
      "section": "Section 1 — Strategy, Vision & Technical Leadership",
      "question": "What's your position on maintaining a proprietary model vs. always using third-party APIs?",
      "difficulty": 2
    },
    {
      "id": 19,
      "section": "Section 1 — Strategy, Vision & Technical Leadership",
      "question": "How would you structure a build to make your architecture resilient to a single model provider's outage?",
      "difficulty": 3
    },
    {
      "id": 20,
      "section": "Section 1 — Strategy, Vision & Technical Leadership",
      "question": "How do you handle a CEO mandate to \"add AI everywhere\" without diluting quality?",
      "difficulty": 3
    },
    {
      "id": 21,
      "section": "Section 1 — Strategy, Vision & Technical Leadership",
      "question": "Describe your approach to technical due diligence when acquiring an AI-heavy startup.",
      "difficulty": 2
    },
    {
      "id": 22,
      "section": "Section 1 — Strategy, Vision & Technical Leadership",
      "question": "How do you decide the right amount of standardization vs. team autonomy in tool choice?",
      "difficulty": 2
    },
    {
      "id": 23,
      "section": "Section 1 — Strategy, Vision & Technical Leadership",
      "question": "What signals tell you an AI initiative should be killed rather than iterated on?",
      "difficulty": 3
    },
    {
      "id": 24,
      "section": "Section 1 — Strategy, Vision & Technical Leadership",
      "question": "How do you plan compute capacity (GPU/TPU) 12 months ahead under uncertain demand?",
      "difficulty": 3
    },
    {
      "id": 25,
      "section": "Section 1 — Strategy, Vision & Technical Leadership",
      "question": "How would you pitch a multi-year AI infrastructure investment to a skeptical CFO?",
      "difficulty": 3
    },
    {
      "id": 26,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "Tell me about a time you said no to a stakeholder's AI feature request.",
      "difficulty": 2
    },
    {
      "id": 27,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "Describe a time an ML/AI project failed — what did you learn?",
      "difficulty": 2
    },
    {
      "id": 28,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "How do you mentor engineers strong in software but new to ML/AI?",
      "difficulty": 3
    },
    {
      "id": 29,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "How do you resolve a technical disagreement between two senior engineers on architecture?",
      "difficulty": 2
    },
    {
      "id": 30,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "How do you influence a roadmap without direct authority over the teams involved?",
      "difficulty": 2
    },
    {
      "id": 31,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "Describe balancing research exploration against shipping deadlines.",
      "difficulty": 3
    },
    {
      "id": 32,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "How do you evangelize AI literacy across a non-technical leadership team?",
      "difficulty": 2
    },
    {
      "id": 33,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "Tell me about an irreversible architectural decision you made with incomplete information.",
      "difficulty": 3
    },
    {
      "id": 34,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "Describe a time you had to deliver bad news about an AI project's timeline or feasibility.",
      "difficulty": 3
    },
    {
      "id": 35,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "Tell me about a time you changed your mind on a technical stance after being challenged.",
      "difficulty": 3
    },
    {
      "id": 36,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "How do you handle an engineer who consistently overpromises on model performance?",
      "difficulty": 2
    },
    {
      "id": 37,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "Describe how you've built psychological safety on a team shipping experimental AI features.",
      "difficulty": 3
    },
    {
      "id": 38,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "Tell me about a conflict between the AI/ML team and a product team over model behavior.",
      "difficulty": 2
    },
    {
      "id": 39,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "How do you run a postmortem after a public AI failure (biased output, hallucination, outage)?",
      "difficulty": 3
    },
    {
      "id": 40,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "Describe how you've hired for an AI team — what do you screen for beyond technical skill?",
      "difficulty": 2
    },
    {
      "id": 41,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "How do you handle attrition of a key AI engineer mid-project?",
      "difficulty": 3
    },
    {
      "id": 42,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "Tell me about a time data or compute constraints forced you to change your architecture.",
      "difficulty": 3
    },
    {
      "id": 43,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "Describe how you've communicated uncertainty in a model's predictions to a non-technical stakeholder.",
      "difficulty": 2
    },
    {
      "id": 44,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "How do you decide when to escalate a disagreement rather than resolve it at your level?",
      "difficulty": 2
    },
    {
      "id": 45,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "Tell me about giving critical feedback to a peer or senior leader on their technical proposal.",
      "difficulty": 2
    },
    {
      "id": 46,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "How do you build trust with legal/compliance teams skeptical of GenAI?",
      "difficulty": 2
    },
    {
      "id": 47,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "Describe a time you had to push back on unrealistic model-accuracy expectations from leadership.",
      "difficulty": 3
    },
    {
      "id": 48,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "How do you manage a cross-functional team spanning data science, platform, and product?",
      "difficulty": 2
    },
    {
      "id": 49,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "Tell me about a time you delegated a high-stakes architectural decision — how did you set them up to succeed?",
      "difficulty": 2
    },
    {
      "id": 50,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "Describe how you handle scope creep on an AI project driven by stakeholder excitement.",
      "difficulty": 2
    },
    {
      "id": 51,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "How do you decide which technical debt to pay down vs. defer on an AI platform?",
      "difficulty": 3
    },
    {
      "id": 52,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "Tell me about a time you advocated for slowing down a launch for safety/quality reasons.",
      "difficulty": 3
    },
    {
      "id": 53,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "How do you build consensus across teams with conflicting incentives on shared AI infrastructure?",
      "difficulty": 3
    },
    {
      "id": 54,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "Describe how you onboard a new engineer into a complex, fast-moving AI codebase.",
      "difficulty": 3
    },
    {
      "id": 55,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "Tell me about a time you had to learn a new domain quickly to lead an AI initiative.",
      "difficulty": 2
    },
    {
      "id": 56,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "How do you keep a team motivated during a long, uncertain research-heavy project?",
      "difficulty": 3
    },
    {
      "id": 57,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "Describe how you've handled a vendor/provider relationship going wrong (price hike, deprecation, outage).",
      "difficulty": 3
    },
    {
      "id": 58,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "Tell me about a time you had to balance innovation with regulatory constraints.",
      "difficulty": 2
    },
    {
      "id": 59,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "How do you decide which metrics to report upward vs. which stay internal to the team?",
      "difficulty": 3
    },
    {
      "id": 60,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "Describe a time you identified a risk in an AI system before it became a problem.",
      "difficulty": 2
    },
    {
      "id": 61,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "How do you structure 1:1s differently for ML researchers vs. platform engineers?",
      "difficulty": 2
    },
    {
      "id": 62,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "Tell me about your proudest technical achievement leading an AI team.",
      "difficulty": 2
    },
    {
      "id": 63,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "Describe how you handle disagreement with your own manager on AI strategy.",
      "difficulty": 2
    },
    {
      "id": 64,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "How do you decide when to bring in outside consultants/vendors vs. build internal expertise?",
      "difficulty": 3
    },
    {
      "id": 65,
      "section": "Section 2 — Leadership & Behavioral",
      "question": "What's a belief about AI systems you've changed your mind about in the last two years?",
      "difficulty": 3
    },
    {
      "id": 66,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Explain the bias-variance tradeoff and give an example of each extreme.",
      "difficulty": 2
    },
    {
      "id": 67,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Supervised vs. unsupervised vs. semi-supervised vs. reinforcement learning — differences and examples.",
      "difficulty": 1
    },
    {
      "id": 68,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Explain linear regression's assumptions and what breaks when they're violated.",
      "difficulty": 2
    },
    {
      "id": 69,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Explain logistic regression and why it uses log-loss rather than MSE.",
      "difficulty": 1
    },
    {
      "id": 70,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "What is regularization? Compare L1 (Lasso) vs. L2 (Ridge) vs. Elastic Net.",
      "difficulty": 1
    },
    {
      "id": 71,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Explain gradient descent vs. stochastic gradient descent vs. mini-batch gradient descent.",
      "difficulty": 1
    },
    {
      "id": 72,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "What causes vanishing/exploding gradients and how do you mitigate them?",
      "difficulty": 2
    },
    {
      "id": 73,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Explain decision trees: splitting criteria (Gini vs. entropy/information gain).",
      "difficulty": 2
    },
    {
      "id": 74,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "How does pruning work in decision trees, and why does it matter?",
      "difficulty": 2
    },
    {
      "id": 75,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Explain bagging vs. boosting; how does Random Forest differ from XGBoost?",
      "difficulty": 1
    },
    {
      "id": 76,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Explain the mechanics of gradient boosting (residual fitting).",
      "difficulty": 1
    },
    {
      "id": 77,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Compare AdaBoost, Gradient Boosting, and XGBoost/LightGBM/CatBoost.",
      "difficulty": 2
    },
    {
      "id": 78,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "What is the kernel trick in SVMs, and when would you use RBF vs. polynomial vs. linear kernels?",
      "difficulty": 1
    },
    {
      "id": 79,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Compare Perceptron and SVM.",
      "difficulty": 2
    },
    {
      "id": 80,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Explain k-Nearest Neighbors — how do you choose k, and what's the curse of dimensionality's effect?",
      "difficulty": 2
    },
    {
      "id": 81,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Difference between KNN and K-Means.",
      "difficulty": 2
    },
    {
      "id": 82,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Explain Naive Bayes and why the \"naive\" independence assumption still works well in practice.",
      "difficulty": 1
    },
    {
      "id": 83,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "What is Maximum Likelihood Estimation, and how does it relate to loss functions?",
      "difficulty": 2
    },
    {
      "id": 84,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Explain confusion matrix, precision, recall, F1, and when you'd optimize for each.",
      "difficulty": 1
    },
    {
      "id": 85,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Explain ROC-AUC vs. PR-AUC — when is PR-AUC more informative?",
      "difficulty": 1
    },
    {
      "id": 86,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Type I vs. Type II error — give a business example of when each is more costly.",
      "difficulty": 2
    },
    {
      "id": 87,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Explain class imbalance and techniques to address it (SMOTE, oversampling, undersampling, class weights).",
      "difficulty": 2
    },
    {
      "id": 88,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "What is cross-validation, and how does k-fold differ from stratified k-fold?",
      "difficulty": 2
    },
    {
      "id": 89,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Explain overfitting vs. underfitting and concrete mitigation techniques for each.",
      "difficulty": 2
    },
    {
      "id": 90,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Explain feature selection vs. feature extraction, with methods for each.",
      "difficulty": 1
    },
    {
      "id": 91,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "What is PCA, mathematically, and when does it fail?",
      "difficulty": 1
    },
    {
      "id": 92,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Compare PCA, t-SNE, UMAP, and autoencoders for dimensionality reduction.",
      "difficulty": 2
    },
    {
      "id": 93,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Explain Linear Discriminant Analysis and how it differs from PCA.",
      "difficulty": 1
    },
    {
      "id": 94,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "What is multicollinearity, and how do you detect and address it?",
      "difficulty": 1
    },
    {
      "id": 95,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Explain the difference between correlation and covariance.",
      "difficulty": 1
    },
    {
      "id": 96,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "What is ANOVA, and when would you use it over a t-test?",
      "difficulty": 1
    },
    {
      "id": 97,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Explain hypothesis testing: null/alternative hypotheses, p-values, significance level.",
      "difficulty": 1
    },
    {
      "id": 98,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "What is a Z-score, and how is it used for outlier detection?",
      "difficulty": 2
    },
    {
      "id": 99,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Explain IQR-based outlier detection and its limits.",
      "difficulty": 1
    },
    {
      "id": 100,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "What sampling techniques exist (simple random, stratified, cluster, systematic, multistage)?",
      "difficulty": 2
    },
    {
      "id": 101,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Explain ensemble learning broadly — why do ensembles typically outperform single models?",
      "difficulty": 2
    },
    {
      "id": 102,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "What is stacking, and how does it differ from bagging/boosting?",
      "difficulty": 2
    },
    {
      "id": 103,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Explain the exploration-exploitation tradeoff in reinforcement learning.",
      "difficulty": 2
    },
    {
      "id": 104,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "What's the difference between model-based and model-free RL?",
      "difficulty": 1
    },
    {
      "id": 105,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Explain Markov Decision Processes and the Bellman equation at a high level.",
      "difficulty": 1
    },
    {
      "id": 106,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "What is Q-learning, and how does Deep Q-Networks extend it?",
      "difficulty": 2
    },
    {
      "id": 107,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Explain policy gradient methods vs. value-based RL methods.",
      "difficulty": 2
    },
    {
      "id": 108,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "What is multi-armed bandit, and when would you use it over full RL?",
      "difficulty": 1
    },
    {
      "id": 109,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Explain collaborative filtering vs. content-based filtering in recommender systems.",
      "difficulty": 2
    },
    {
      "id": 110,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "What is matrix factorization, and how does it apply to recommendation?",
      "difficulty": 2
    },
    {
      "id": 111,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Explain cold-start problems in recommender systems and mitigation strategies.",
      "difficulty": 1
    },
    {
      "id": 112,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "What's the difference between explicit and implicit feedback in recsys?",
      "difficulty": 2
    },
    {
      "id": 113,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Explain the exposure/popularity bias problem in recommendation and how to correct for it.",
      "difficulty": 1
    },
    {
      "id": 114,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "What is calibration in ML models, and why does it matter for probabilistic predictions?",
      "difficulty": 2
    },
    {
      "id": 115,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Explain the difference between generative and discriminative models.",
      "difficulty": 1
    },
    {
      "id": 116,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "What is the EM (Expectation-Maximization) algorithm used for?",
      "difficulty": 1
    },
    {
      "id": 117,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Explain Gaussian Mixture Models vs. K-Means clustering.",
      "difficulty": 2
    },
    {
      "id": 118,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "What is hierarchical clustering, and how do you choose the number of clusters?",
      "difficulty": 1
    },
    {
      "id": 119,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Explain DBSCAN and when density-based clustering beats K-Means.",
      "difficulty": 1
    },
    {
      "id": 120,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "What is the silhouette score, and how do you use it to evaluate clustering?",
      "difficulty": 2
    },
    {
      "id": 121,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Explain survival analysis and when it applies over standard regression/classification.",
      "difficulty": 1
    },
    {
      "id": 122,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "What is A/B testing, and how do you determine statistical significance and sample size?",
      "difficulty": 1
    },
    {
      "id": 123,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Explain multi-armed bandit approaches to A/B testing vs. fixed-horizon testing.",
      "difficulty": 2
    },
    {
      "id": 124,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "What is Simpson's Paradox, and how could it mislead an experiment analysis?",
      "difficulty": 2
    },
    {
      "id": 125,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Explain the difference between causal inference and correlation-based ML.",
      "difficulty": 1
    },
    {
      "id": 126,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "What is propensity score matching, and when would you use it?",
      "difficulty": 1
    },
    {
      "id": 127,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Explain uplift modeling and how it differs from standard response modeling.",
      "difficulty": 2
    },
    {
      "id": 128,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "What is feature leakage, and how do you detect it before it silently inflates metrics?",
      "difficulty": 2
    },
    {
      "id": 129,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Explain train/validation/test split strategy for time-dependent data.",
      "difficulty": 1
    },
    {
      "id": 130,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "What is target/mean encoding, and what risk does it carry?",
      "difficulty": 1
    },
    {
      "id": 131,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Explain one-hot encoding vs. embedding-based categorical encoding, and when each is appropriate.",
      "difficulty": 1
    },
    {
      "id": 132,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "What is weight decay, and how does it relate to L2 regularization?",
      "difficulty": 1
    },
    {
      "id": 133,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Explain early stopping as a regularization technique.",
      "difficulty": 1
    },
    {
      "id": 134,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "What is the difference between parametric and non-parametric models?",
      "difficulty": 2
    },
    {
      "id": 135,
      "section": "Section 3 — Classic ML Fundamentals",
      "question": "Explain how you would build a churn-prediction model end to end.",
      "difficulty": 1
    },
    {
      "id": 136,
      "section": "Section 4 — Statistics & Probability",
      "question": "Explain conditional probability and Bayes' Theorem with an example.",
      "difficulty": 1
    },
    {
      "id": 137,
      "section": "Section 4 — Statistics & Probability",
      "question": "What's the difference between joint, marginal, and conditional probability?",
      "difficulty": 1
    },
    {
      "id": 138,
      "section": "Section 4 — Statistics & Probability",
      "question": "Explain the Central Limit Theorem and why it matters for ML.",
      "difficulty": 2
    },
    {
      "id": 139,
      "section": "Section 4 — Statistics & Probability",
      "question": "What is a p-value, and what's the most common misinterpretation of it?",
      "difficulty": 1
    },
    {
      "id": 140,
      "section": "Section 4 — Statistics & Probability",
      "question": "Explain Type I vs. Type II error in the context of hypothesis testing (not just classification).",
      "difficulty": 2
    },
    {
      "id": 141,
      "section": "Section 4 — Statistics & Probability",
      "question": "What is a confidence interval, and how do you interpret a 95% CI correctly?",
      "difficulty": 2
    },
    {
      "id": 142,
      "section": "Section 4 — Statistics & Probability",
      "question": "Explain the difference between population and sample statistics.",
      "difficulty": 1
    },
    {
      "id": 143,
      "section": "Section 4 — Statistics & Probability",
      "question": "What is KL divergence, and where does it show up in ML (VAEs, RLHF, distillation)?",
      "difficulty": 1
    },
    {
      "id": 144,
      "section": "Section 4 — Statistics & Probability",
      "question": "Explain cross-entropy and its relationship to KL divergence.",
      "difficulty": 2
    },
    {
      "id": 145,
      "section": "Section 4 — Statistics & Probability",
      "question": "What is entropy in information theory, and how does it relate to decision tree splits?",
      "difficulty": 2
    },
    {
      "id": 146,
      "section": "Section 4 — Statistics & Probability",
      "question": "Explain the law of large numbers vs. the Central Limit Theorem.",
      "difficulty": 2
    },
    {
      "id": 147,
      "section": "Section 4 — Statistics & Probability",
      "question": "What distribution would you use to model event counts over time, and why (Poisson)?",
      "difficulty": 2
    },
    {
      "id": 148,
      "section": "Section 4 — Statistics & Probability",
      "question": "Explain the difference between a binomial and multinomial distribution.",
      "difficulty": 2
    },
    {
      "id": 149,
      "section": "Section 4 — Statistics & Probability",
      "question": "What is a normal distribution's role in statistical modeling, and when is it a poor assumption?",
      "difficulty": 1
    },
    {
      "id": 150,
      "section": "Section 4 — Statistics & Probability",
      "question": "Explain skewness and kurtosis, and how they'd change your modeling approach.",
      "difficulty": 1
    },
    {
      "id": 151,
      "section": "Section 4 — Statistics & Probability",
      "question": "What is bootstrapping, and how is it used to estimate uncertainty?",
      "difficulty": 1
    },
    {
      "id": 152,
      "section": "Section 4 — Statistics & Probability",
      "question": "Explain the difference between frequentist and Bayesian statistics.",
      "difficulty": 1
    },
    {
      "id": 153,
      "section": "Section 4 — Statistics & Probability",
      "question": "What is a prior, likelihood, and posterior in Bayesian inference?",
      "difficulty": 2
    },
    {
      "id": 154,
      "section": "Section 4 — Statistics & Probability",
      "question": "Explain Markov Chain Monte Carlo (MCMC) at a conceptual level.",
      "difficulty": 1
    },
    {
      "id": 155,
      "section": "Section 4 — Statistics & Probability",
      "question": "What is the difference between correlation and causation, with an example of confounding?",
      "difficulty": 1
    },
    {
      "id": 156,
      "section": "Section 4 — Statistics & Probability",
      "question": "Explain multiple hypothesis testing and the need for correction (Bonferroni, FDR).",
      "difficulty": 1
    },
    {
      "id": 157,
      "section": "Section 4 — Statistics & Probability",
      "question": "What is heteroscedasticity, and why does it matter for regression models?",
      "difficulty": 1
    },
    {
      "id": 158,
      "section": "Section 4 — Statistics & Probability",
      "question": "Explain autocorrelation and why it matters for time-series regression.",
      "difficulty": 1
    },
    {
      "id": 159,
      "section": "Section 4 — Statistics & Probability",
      "question": "What is stationarity in time series, and how do you test for it?",
      "difficulty": 1
    },
    {
      "id": 160,
      "section": "Section 4 — Statistics & Probability",
      "question": "Explain variance inflation factor (VIF) and multicollinearity detection.",
      "difficulty": 1
    },
    {
      "id": 161,
      "section": "Section 4 — Statistics & Probability",
      "question": "What's the difference between a t-test and a chi-squared test — when do you use each?",
      "difficulty": 1
    },
    {
      "id": 162,
      "section": "Section 4 — Statistics & Probability",
      "question": "Explain the difference between one-tailed and two-tailed tests.",
      "difficulty": 1
    },
    {
      "id": 163,
      "section": "Section 4 — Statistics & Probability",
      "question": "What is Bayesian A/B testing, and how does it differ from frequentist A/B testing?",
      "difficulty": 2
    },
    {
      "id": 164,
      "section": "Section 4 — Statistics & Probability",
      "question": "Explain regression to the mean and a real business scenario where it misleads decisions.",
      "difficulty": 1
    },
    {
      "id": 165,
      "section": "Section 4 — Statistics & Probability",
      "question": "What is the difference between MLE and MAP estimation?",
      "difficulty": 1
    },
    {
      "id": 166,
      "section": "Section 4 — Statistics & Probability",
      "question": "Explain the concept of a sufficient statistic.",
      "difficulty": 2
    },
    {
      "id": 167,
      "section": "Section 4 — Statistics & Probability",
      "question": "What is the delta method used for in statistics?",
      "difficulty": 2
    },
    {
      "id": 168,
      "section": "Section 4 — Statistics & Probability",
      "question": "Explain power analysis and how it informs experiment design.",
      "difficulty": 1
    },
    {
      "id": 169,
      "section": "Section 4 — Statistics & Probability",
      "question": "What is survivorship bias, and how could it corrupt a training dataset?",
      "difficulty": 1
    },
    {
      "id": 170,
      "section": "Section 4 — Statistics & Probability",
      "question": "Explain Simpson's paradox with a concrete numeric example.",
      "difficulty": 2
    },
    {
      "id": 171,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "What is a neuron in an ANN, mathematically?",
      "difficulty": 1
    },
    {
      "id": 172,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "Explain forward propagation and backpropagation end to end.",
      "difficulty": 2
    },
    {
      "id": 173,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "Why do we need non-linear activation functions?",
      "difficulty": 2
    },
    {
      "id": 174,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "Compare Sigmoid, Tanh, ReLU, Leaky ReLU, GELU, and Swish — tradeoffs?",
      "difficulty": 1
    },
    {
      "id": 175,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "Explain the vanishing gradient problem and how ReLU/residual connections address it.",
      "difficulty": 1
    },
    {
      "id": 176,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "What is batch normalization, and why does it stabilize training?",
      "difficulty": 1
    },
    {
      "id": 177,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "Compare batch norm, layer norm, and group norm — when is each preferred?",
      "difficulty": 2
    },
    {
      "id": 178,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "Explain dropout and why it prevents overfitting.",
      "difficulty": 2
    },
    {
      "id": 179,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "What is weight initialization's role, and compare Xavier/Glorot vs. He initialization.",
      "difficulty": 2
    },
    {
      "id": 180,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "Explain the Adam optimizer and how it differs from vanilla SGD with momentum.",
      "difficulty": 2
    },
    {
      "id": 181,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "What is learning rate scheduling, and name common strategies (cosine, step decay, warmup).",
      "difficulty": 2
    },
    {
      "id": 182,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "Explain gradient clipping and when it's necessary.",
      "difficulty": 1
    },
    {
      "id": 183,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "What is a convolutional layer, and how does weight sharing reduce parameters?",
      "difficulty": 1
    },
    {
      "id": 184,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "Explain pooling layers (max vs. average) and their purpose.",
      "difficulty": 1
    },
    {
      "id": 185,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "What is a receptive field in a CNN, and how does depth affect it?",
      "difficulty": 2
    },
    {
      "id": 186,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "Explain padding and stride in convolutions.",
      "difficulty": 2
    },
    {
      "id": 187,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "What is a residual/skip connection, and why does ResNet train so much deeper than plain CNNs?",
      "difficulty": 1
    },
    {
      "id": 188,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "Explain the architecture and purpose of an autoencoder.",
      "difficulty": 1
    },
    {
      "id": 189,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "What's the difference between an autoencoder and a variational autoencoder (VAE)?",
      "difficulty": 1
    },
    {
      "id": 190,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "Explain GANs: generator, discriminator, and the adversarial training objective.",
      "difficulty": 1
    },
    {
      "id": 191,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "What is mode collapse in GANs, and how do you mitigate it?",
      "difficulty": 2
    },
    {
      "id": 192,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "Explain recurrent neural networks and the vanishing gradient problem specific to RNNs.",
      "difficulty": 1
    },
    {
      "id": 193,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "What is an LSTM, and how do its gates (forget, input, output) solve RNN limitations?",
      "difficulty": 2
    },
    {
      "id": 194,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "Compare LSTM and GRU — what's the tradeoff?",
      "difficulty": 1
    },
    {
      "id": 195,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "Explain sequence-to-sequence models and where they were used before transformers.",
      "difficulty": 2
    },
    {
      "id": 196,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "What is teacher forcing in sequence models, and what problem can it cause at inference time?",
      "difficulty": 2
    },
    {
      "id": 197,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "Explain the concept of attention before transformers (Bahdanau/Luong attention).",
      "difficulty": 1
    },
    {
      "id": 198,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "What is transfer learning, and how does fine-tuning differ from feature extraction?",
      "difficulty": 1
    },
    {
      "id": 199,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "Explain data augmentation techniques for images and why they help generalization.",
      "difficulty": 2
    },
    {
      "id": 200,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "What is knowledge distillation, and how does a student model learn from a teacher?",
      "difficulty": 1
    },
    {
      "id": 201,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "Explain quantization (INT8/INT4) and the accuracy/latency tradeoff.",
      "difficulty": 1
    },
    {
      "id": 202,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "What is pruning in neural networks, and how does structured differ from unstructured pruning?",
      "difficulty": 1
    },
    {
      "id": 203,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "Explain the universal approximation theorem and its practical limitations.",
      "difficulty": 1
    },
    {
      "id": 204,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "What is catastrophic forgetting, and how do continual learning methods address it?",
      "difficulty": 1
    },
    {
      "id": 205,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "Explain the difference between epoch, batch, and iteration.",
      "difficulty": 1
    },
    {
      "id": 206,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "What is curriculum learning?",
      "difficulty": 2
    },
    {
      "id": 207,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "Explain self-supervised learning and give two pretext-task examples.",
      "difficulty": 2
    },
    {
      "id": 208,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "What is contrastive learning (e.g., SimCLR, CLIP) and how does the loss function work?",
      "difficulty": 2
    },
    {
      "id": 209,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "Explain the role of the loss function's curvature in optimization difficulty (saddle points, local minima).",
      "difficulty": 1
    },
    {
      "id": 210,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "What is label smoothing, and why does it help calibration?",
      "difficulty": 2
    },
    {
      "id": 211,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "Explain mixed-precision training and why it speeds up training without hurting accuracy much.",
      "difficulty": 1
    },
    {
      "id": 212,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "What is gradient checkpointing, and what tradeoff does it make?",
      "difficulty": 1
    },
    {
      "id": 213,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "Explain data parallelism vs. model parallelism vs. pipeline parallelism in distributed training.",
      "difficulty": 2
    },
    {
      "id": 214,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "What is a loss landscape, and how does it relate to generalization?",
      "difficulty": 1
    },
    {
      "id": 215,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "Explain the exploding gradient problem and gradient clipping as a fix.",
      "difficulty": 2
    },
    {
      "id": 216,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "What is weight tying, and where is it used (e.g., embedding/output layer sharing)?",
      "difficulty": 2
    },
    {
      "id": 217,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "Explain the difference between online learning and batch learning.",
      "difficulty": 2
    },
    {
      "id": 218,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "What is few-shot learning, and how does it differ from zero-shot learning?",
      "difficulty": 2
    },
    {
      "id": 219,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "Explain meta-learning (\"learning to learn\") at a conceptual level.",
      "difficulty": 2
    },
    {
      "id": 220,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "What is neural architecture search (NAS)?",
      "difficulty": 2
    },
    {
      "id": 221,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "Explain the difference between generative and discriminative deep learning models.",
      "difficulty": 1
    },
    {
      "id": 222,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "What is a Siamese network, and where is it used (e.g., face verification)?",
      "difficulty": 1
    },
    {
      "id": 223,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "Explain triplet loss and its role in embedding learning.",
      "difficulty": 2
    },
    {
      "id": 224,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "What is the role of temperature in softmax outputs?",
      "difficulty": 1
    },
    {
      "id": 225,
      "section": "Section 5 — Deep Learning Fundamentals",
      "question": "Explain why deeper networks generally outperform wider shallow ones, and where that breaks down.",
      "difficulty": 1
    },
    {
      "id": 226,
      "section": "Section 6 — Computer Vision",
      "question": "Explain image classification vs. object detection vs. semantic segmentation vs. instance segmentation.",
      "difficulty": 1
    },
    {
      "id": 227,
      "section": "Section 6 — Computer Vision",
      "question": "Compare two-stage detectors (Faster R-CNN) vs. one-stage detectors (YOLO, SSD).",
      "difficulty": 2
    },
    {
      "id": 228,
      "section": "Section 6 — Computer Vision",
      "question": "What is non-max suppression, and why is it needed in object detection?",
      "difficulty": 1
    },
    {
      "id": 229,
      "section": "Section 6 — Computer Vision",
      "question": "Explain anchor boxes and their role in detection models.",
      "difficulty": 1
    },
    {
      "id": 230,
      "section": "Section 6 — Computer Vision",
      "question": "What is IoU (Intersection over Union), and how is it used in evaluation?",
      "difficulty": 2
    },
    {
      "id": 231,
      "section": "Section 6 — Computer Vision",
      "question": "Explain mAP (mean Average Precision) as an object-detection metric.",
      "difficulty": 1
    },
    {
      "id": 232,
      "section": "Section 6 — Computer Vision",
      "question": "What is a feature pyramid network, and why does it help detect objects at multiple scales?",
      "difficulty": 1
    },
    {
      "id": 233,
      "section": "Section 6 — Computer Vision",
      "question": "Explain image segmentation approaches: thresholding, U-Net, Mask R-CNN.",
      "difficulty": 1
    },
    {
      "id": 234,
      "section": "Section 6 — Computer Vision",
      "question": "What is optical flow, and where is it used in video understanding?",
      "difficulty": 1
    },
    {
      "id": 235,
      "section": "Section 6 — Computer Vision",
      "question": "Explain pose estimation and common architectures (OpenPose, HRNet).",
      "difficulty": 1
    },
    {
      "id": 236,
      "section": "Section 6 — Computer Vision",
      "question": "What is style transfer, and how do content/style losses work?",
      "difficulty": 1
    },
    {
      "id": 237,
      "section": "Section 6 — Computer Vision",
      "question": "Explain image captioning architectures combining CNN encoders and language decoders.",
      "difficulty": 1
    },
    {
      "id": 238,
      "section": "Section 6 — Computer Vision",
      "question": "What is OCR, and how do modern OCR pipelines differ from classic ones?",
      "difficulty": 2
    },
    {
      "id": 239,
      "section": "Section 6 — Computer Vision",
      "question": "Explain Vision Transformers (ViT) and how patch embeddings replace convolutions.",
      "difficulty": 1
    },
    {
      "id": 240,
      "section": "Section 6 — Computer Vision",
      "question": "Compare CNNs and ViTs — when does each perform better, and why?",
      "difficulty": 1
    },
    {
      "id": 241,
      "section": "Section 6 — Computer Vision",
      "question": "What is CLIP, and how does contrastive image-text pretraining work?",
      "difficulty": 1
    },
    {
      "id": 242,
      "section": "Section 6 — Computer Vision",
      "question": "Explain diffusion models for image generation at a conceptual level.",
      "difficulty": 1
    },
    {
      "id": 243,
      "section": "Section 6 — Computer Vision",
      "question": "Compare GANs and diffusion models for image synthesis — tradeoffs?",
      "difficulty": 2
    },
    {
      "id": 244,
      "section": "Section 6 — Computer Vision",
      "question": "What is super-resolution, and what architectures are commonly used?",
      "difficulty": 2
    },
    {
      "id": 245,
      "section": "Section 6 — Computer Vision",
      "question": "Explain data augmentation strategies specific to vision (mixup, cutmix, random erasing).",
      "difficulty": 2
    },
    {
      "id": 246,
      "section": "Section 6 — Computer Vision",
      "question": "What is domain adaptation in computer vision, and why does it matter for deployment?",
      "difficulty": 1
    },
    {
      "id": 247,
      "section": "Section 6 — Computer Vision",
      "question": "Explain few-shot object detection challenges and approaches.",
      "difficulty": 2
    },
    {
      "id": 248,
      "section": "Section 6 — Computer Vision",
      "question": "What is 3D computer vision (point clouds, depth estimation), and how does it differ from 2D?",
      "difficulty": 1
    },
    {
      "id": 249,
      "section": "Section 6 — Computer Vision",
      "question": "Explain video understanding architectures (3D CNNs, video transformers).",
      "difficulty": 2
    },
    {
      "id": 250,
      "section": "Section 6 — Computer Vision",
      "question": "What is face recognition's typical pipeline (detection, alignment, embedding, matching)?",
      "difficulty": 2
    },
    {
      "id": 251,
      "section": "Section 6 — Computer Vision",
      "question": "Explain adversarial examples in computer vision and their implications for production systems.",
      "difficulty": 1
    },
    {
      "id": 252,
      "section": "Section 6 — Computer Vision",
      "question": "What is image inpainting, and what architectures are used?",
      "difficulty": 2
    },
    {
      "id": 253,
      "section": "Section 6 — Computer Vision",
      "question": "Explain multimodal vision-language models and how image tokens are fed into an LLM.",
      "difficulty": 2
    },
    {
      "id": 254,
      "section": "Section 6 — Computer Vision",
      "question": "What is a scene graph, and where is it used?",
      "difficulty": 2
    },
    {
      "id": 255,
      "section": "Section 6 — Computer Vision",
      "question": "Explain the tradeoffs of on-device (edge) vs. cloud inference for vision models.",
      "difficulty": 1
    },
    {
      "id": 256,
      "section": "Section 7 — NLP Fundamentals (Pre-LLM)",
      "question": "Explain tokenization approaches: word-level, character-level, subword (BPE, WordPiece, SentencePiece).",
      "difficulty": 1
    },
    {
      "id": 257,
      "section": "Section 7 — NLP Fundamentals (Pre-LLM)",
      "question": "What is TF-IDF, and what are its limitations compared to embeddings?",
      "difficulty": 2
    },
    {
      "id": 258,
      "section": "Section 7 — NLP Fundamentals (Pre-LLM)",
      "question": "Explain word2vec — CBOW vs. skip-gram.",
      "difficulty": 1
    },
    {
      "id": 259,
      "section": "Section 7 — NLP Fundamentals (Pre-LLM)",
      "question": "What is GloVe, and how does it differ from word2vec?",
      "difficulty": 1
    },
    {
      "id": 260,
      "section": "Section 7 — NLP Fundamentals (Pre-LLM)",
      "question": "Explain the difference between static embeddings and contextual embeddings (ELMo, BERT).",
      "difficulty": 1
    },
    {
      "id": 261,
      "section": "Section 7 — NLP Fundamentals (Pre-LLM)",
      "question": "What is Named Entity Recognition, and what architectures were used pre-transformer (CRF, BiLSTM-CRF)?",
      "difficulty": 2
    },
    {
      "id": 262,
      "section": "Section 7 — NLP Fundamentals (Pre-LLM)",
      "question": "Explain part-of-speech tagging and its role in NLP pipelines.",
      "difficulty": 1
    },
    {
      "id": 263,
      "section": "Section 7 — NLP Fundamentals (Pre-LLM)",
      "question": "What is dependency parsing vs. constituency parsing?",
      "difficulty": 2
    },
    {
      "id": 264,
      "section": "Section 7 — NLP Fundamentals (Pre-LLM)",
      "question": "Explain topic modeling (LDA) at a conceptual level.",
      "difficulty": 1
    },
    {
      "id": 265,
      "section": "Section 7 — NLP Fundamentals (Pre-LLM)",
      "question": "What is sentiment analysis, and what challenges arise with sarcasm/negation?",
      "difficulty": 1
    },
    {
      "id": 266,
      "section": "Section 7 — NLP Fundamentals (Pre-LLM)",
      "question": "Explain n-gram language models and their limitations vs. neural language models.",
      "difficulty": 2
    },
    {
      "id": 267,
      "section": "Section 7 — NLP Fundamentals (Pre-LLM)",
      "question": "What is perplexity, and how is it used to evaluate language models?",
      "difficulty": 2
    },
    {
      "id": 268,
      "section": "Section 7 — NLP Fundamentals (Pre-LLM)",
      "question": "Explain BLEU, ROUGE, and METEOR — what do they measure and where do they fall short?",
      "difficulty": 1
    },
    {
      "id": 269,
      "section": "Section 7 — NLP Fundamentals (Pre-LLM)",
      "question": "What is text classification, and what are common architectures pre-transformer (CNN-text, BiLSTM)?",
      "difficulty": 2
    },
    {
      "id": 270,
      "section": "Section 7 — NLP Fundamentals (Pre-LLM)",
      "question": "Explain coreference resolution and why it's hard.",
      "difficulty": 2
    },
    {
      "id": 271,
      "section": "Section 7 — NLP Fundamentals (Pre-LLM)",
      "question": "What is machine translation's evolution from statistical MT to seq2seq to transformer-based MT?",
      "difficulty": 1
    },
    {
      "id": 272,
      "section": "Section 7 — NLP Fundamentals (Pre-LLM)",
      "question": "Explain the concept of word sense disambiguation.",
      "difficulty": 2
    },
    {
      "id": 273,
      "section": "Section 7 — NLP Fundamentals (Pre-LLM)",
      "question": "What is text summarization — extractive vs. abstractive — and give an architecture for each.",
      "difficulty": 1
    },
    {
      "id": 274,
      "section": "Section 7 — NLP Fundamentals (Pre-LLM)",
      "question": "Explain stemming vs. lemmatization.",
      "difficulty": 1
    },
    {
      "id": 275,
      "section": "Section 7 — NLP Fundamentals (Pre-LLM)",
      "question": "What is stopword removal, and when might removing stopwords hurt rather than help?",
      "difficulty": 2
    },
    {
      "id": 276,
      "section": "Section 7 — NLP Fundamentals (Pre-LLM)",
      "question": "Explain the bag-of-words model and its limitations.",
      "difficulty": 1
    },
    {
      "id": 277,
      "section": "Section 7 — NLP Fundamentals (Pre-LLM)",
      "question": "What is a language model, fundamentally, and how does perplexity relate to cross-entropy?",
      "difficulty": 1
    },
    {
      "id": 278,
      "section": "Section 7 — NLP Fundamentals (Pre-LLM)",
      "question": "Explain speech recognition's basic pipeline (acoustic model, language model, decoder).",
      "difficulty": 1
    },
    {
      "id": 279,
      "section": "Section 7 — NLP Fundamentals (Pre-LLM)",
      "question": "What is text-to-speech, and how have neural TTS systems (Tacotron, WaveNet) changed the field?",
      "difficulty": 2
    },
    {
      "id": 280,
      "section": "Section 7 — NLP Fundamentals (Pre-LLM)",
      "question": "Explain semantic search vs. keyword/lexical search (BM25).",
      "difficulty": 2
    },
    {
      "id": 281,
      "section": "Section 7 — NLP Fundamentals (Pre-LLM)",
      "question": "What is BM25, and how does it improve on TF-IDF?",
      "difficulty": 1
    },
    {
      "id": 282,
      "section": "Section 7 — NLP Fundamentals (Pre-LLM)",
      "question": "Explain question answering system design pre-LLM (extractive QA with BERT).",
      "difficulty": 2
    },
    {
      "id": 283,
      "section": "Section 7 — NLP Fundamentals (Pre-LLM)",
      "question": "What is entity linking, and how does it connect to knowledge graphs?",
      "difficulty": 1
    },
    {
      "id": 284,
      "section": "Section 7 — NLP Fundamentals (Pre-LLM)",
      "question": "Explain intent classification and slot filling in a traditional dialogue system.",
      "difficulty": 2
    },
    {
      "id": 285,
      "section": "Section 7 — NLP Fundamentals (Pre-LLM)",
      "question": "What is text normalization, and why does it matter for downstream NLP tasks?",
      "difficulty": 2
    },
    {
      "id": 286,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "Explain the transformer architecture end to end (encoder, decoder, attention, feed-forward).",
      "difficulty": 3
    },
    {
      "id": 287,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "Derive/explain scaled dot-product attention and why scaling by √d_k matters.",
      "difficulty": 2
    },
    {
      "id": 288,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "What is multi-head attention, and why use multiple heads instead of one large head?",
      "difficulty": 2
    },
    {
      "id": 289,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "Explain positional encoding — sinusoidal vs. learned vs. rotary (RoPE).",
      "difficulty": 3
    },
    {
      "id": 290,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "What is RoPE, and how does RoPE scaling/YaRN extend context length?",
      "difficulty": 3
    },
    {
      "id": 291,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "Compare Multi-Head Attention (MHA), Multi-Query Attention (MQA), and Grouped-Query Attention (GQA).",
      "difficulty": 2
    },
    {
      "id": 292,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "Why do modern LLMs favor GQA over full MHA?",
      "difficulty": 2
    },
    {
      "id": 293,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "Explain KV cache and why it's essential for efficient autoregressive generation.",
      "difficulty": 3
    },
    {
      "id": 294,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "What is Multi-Head Latent Attention (MLA), and what problem does it solve versus GQA?",
      "difficulty": 2
    },
    {
      "id": 295,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "Explain layer normalization placement (pre-LN vs. post-LN) and its effect on training stability.",
      "difficulty": 3
    },
    {
      "id": 296,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "What is the feed-forward network's role inside a transformer block?",
      "difficulty": 2
    },
    {
      "id": 297,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "Explain the difference between encoder-only, decoder-only, and encoder-decoder transformer architectures.",
      "difficulty": 3
    },
    {
      "id": 298,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "Why do most modern LLMs use decoder-only architectures?",
      "difficulty": 3
    },
    {
      "id": 299,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "Explain masked self-attention and why it's needed for autoregressive generation.",
      "difficulty": 2
    },
    {
      "id": 300,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "What is causal masking, and how does it differ from padding masks?",
      "difficulty": 2
    },
    {
      "id": 301,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "Explain byte-pair encoding (BPE) tokenization and its impact on model behavior with rare words.",
      "difficulty": 2
    },
    {
      "id": 302,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "What is the vocabulary size tradeoff in tokenizer design?",
      "difficulty": 2
    },
    {
      "id": 303,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "Explain pretraining objectives: causal LM (next-token prediction) vs. masked LM (BERT-style).",
      "difficulty": 2
    },
    {
      "id": 304,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "What is the scaling law (Chinchilla-style), and how does it inform compute-optimal training?",
      "difficulty": 3
    },
    {
      "id": 305,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "Explain the difference between model parameters, training tokens, and compute (FLOPs) in scaling laws.",
      "difficulty": 2
    },
    {
      "id": 306,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "What is Mixture-of-Experts (MoE), and how does sparse routing reduce compute per token?",
      "difficulty": 3
    },
    {
      "id": 307,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "Explain load balancing challenges in MoE training.",
      "difficulty": 2
    },
    {
      "id": 308,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "What is the difference between dense and sparse (MoE) LLM architectures in serving cost?",
      "difficulty": 2
    },
    {
      "id": 309,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "Explain SFT (Supervised Fine-Tuning) — what data and objective does it use?",
      "difficulty": 3
    },
    {
      "id": 310,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "What is RLHF, end to end (reward model, PPO, policy)?",
      "difficulty": 3
    },
    {
      "id": 311,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "Explain DPO (Direct Preference Optimization) and how it avoids training a separate reward model.",
      "difficulty": 2
    },
    {
      "id": 312,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "What is GRPO, and how does it differ from PPO in RLHF pipelines?",
      "difficulty": 3
    },
    {
      "id": 313,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "Explain RLVR (Reinforcement Learning from Verifiable Rewards) and where it's used (math, code).",
      "difficulty": 2
    },
    {
      "id": 314,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "What is the role of a reward model in RLHF, and how is it trained?",
      "difficulty": 2
    },
    {
      "id": 315,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "Explain reward hacking in RLHF and how to mitigate it.",
      "difficulty": 2
    },
    {
      "id": 316,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "What is instruction tuning, and how does it differ from RLHF?",
      "difficulty": 3
    },
    {
      "id": 317,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "Explain constitutional AI / AI feedback (RLAIF) as an alternative to human-labeled RLHF.",
      "difficulty": 3
    },
    {
      "id": 318,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "What is LoRA (Low-Rank Adaptation), and why is it parameter-efficient?",
      "difficulty": 2
    },
    {
      "id": 319,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "Compare LoRA, QLoRA, and full fine-tuning — cost/quality tradeoffs.",
      "difficulty": 2
    },
    {
      "id": 320,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "Explain prefix tuning and prompt tuning as PEFT methods.",
      "difficulty": 2
    },
    {
      "id": 321,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "What is catastrophic forgetting during continued pretraining, and how do you prevent it?",
      "difficulty": 2
    },
    {
      "id": 322,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "Explain in-context learning — why can LLMs \"learn\" from examples in the prompt without weight updates?",
      "difficulty": 3
    },
    {
      "id": 323,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "What is emergent behavior in LLMs, and is it a real phenomenon or a measurement artifact (debate both sides)?",
      "difficulty": 2
    },
    {
      "id": 324,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "Explain chain-of-thought reasoning and why it improves performance on multi-step tasks.",
      "difficulty": 2
    },
    {
      "id": 325,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "What is self-consistency decoding, and how does it improve chain-of-thought accuracy?",
      "difficulty": 3
    },
    {
      "id": 326,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "Explain test-time compute / inference-time scaling (reasoning models) and its cost implications.",
      "difficulty": 3
    },
    {
      "id": 327,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "What is a \"thinking budget,\" and how would you tune it for cost vs. accuracy?",
      "difficulty": 2
    },
    {
      "id": 328,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "Explain speculative decoding and why it speeds up inference without changing output distribution.",
      "difficulty": 3
    },
    {
      "id": 329,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "What is continuous batching, and why does it improve GPU utilization for LLM serving?",
      "difficulty": 2
    },
    {
      "id": 330,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "Explain paged attention (vLLM) and how it manages KV cache memory efficiently.",
      "difficulty": 2
    },
    {
      "id": 331,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "What is flash attention, and how does it reduce memory bandwidth bottlenecks?",
      "difficulty": 3
    },
    {
      "id": 332,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "Explain the difference between prefill and decode phases in LLM inference, and why they have different bottlenecks.",
      "difficulty": 2
    },
    {
      "id": 333,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "What is context window, and what architectural/serving factors limit how far it can scale?",
      "difficulty": 3
    },
    {
      "id": 334,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "Explain long-context handling strategies: sliding window attention, sparse attention, retrieval augmentation.",
      "difficulty": 2
    },
    {
      "id": 335,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "What is model distillation for LLMs, and how do you distill a large model into a smaller one?",
      "difficulty": 2
    },
    {
      "id": 336,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "Explain quantization-aware training vs. post-training quantization for LLMs.",
      "difficulty": 3
    },
    {
      "id": 337,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "What is the outlier problem in LLM quantization, and how do techniques like SmoothQuant address it?",
      "difficulty": 3
    },
    {
      "id": 338,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "Explain tokenizer mismatch issues when switching between models or fine-tuning on new domains.",
      "difficulty": 3
    },
    {
      "id": 339,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "What is a system prompt, and how does it differ mechanically from a user prompt?",
      "difficulty": 2
    },
    {
      "id": 340,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "Explain temperature, top-k, and top-p (nucleus) sampling — how do they shape output diversity?",
      "difficulty": 2
    },
    {
      "id": 341,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "What is greedy decoding vs. beam search — tradeoffs for LLM generation?",
      "difficulty": 2
    },
    {
      "id": 342,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "Explain repetition penalty and frequency penalty in decoding.",
      "difficulty": 2
    },
    {
      "id": 343,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "What is model merging (e.g., weight averaging across fine-tunes), and when is it useful?",
      "difficulty": 3
    },
    {
      "id": 344,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "Explain the difference between a base model and an instruct/chat-tuned model.",
      "difficulty": 3
    },
    {
      "id": 345,
      "section": "Section 8 — LLM & Transformer Fundamentals",
      "question": "What is hallucination, mechanistically — why do LLMs generate confident false statements?",
      "difficulty": 3
    },
    {
      "id": 346,
      "section": "Section 9 — Prompt Engineering & Structured Outputs",
      "question": "Explain zero-shot vs. few-shot prompting and when each is appropriate.",
      "difficulty": 2
    },
    {
      "id": 347,
      "section": "Section 9 — Prompt Engineering & Structured Outputs",
      "question": "What is chain-of-thought prompting, and how does \"let's think step by step\" change output quality?",
      "difficulty": 3
    },
    {
      "id": 348,
      "section": "Section 9 — Prompt Engineering & Structured Outputs",
      "question": "Explain ReAct prompting (reasoning + acting) for tool-using agents.",
      "difficulty": 3
    },
    {
      "id": 349,
      "section": "Section 9 — Prompt Engineering & Structured Outputs",
      "question": "What is Tree-of-Thought prompting, and when is it worth the extra inference cost?",
      "difficulty": 3
    },
    {
      "id": 350,
      "section": "Section 9 — Prompt Engineering & Structured Outputs",
      "question": "Explain self-consistency prompting and its cost/accuracy tradeoff.",
      "difficulty": 3
    },
    {
      "id": 351,
      "section": "Section 9 — Prompt Engineering & Structured Outputs",
      "question": "What is prompt chaining, and when should you split one prompt into multiple calls?",
      "difficulty": 2
    },
    {
      "id": 352,
      "section": "Section 9 — Prompt Engineering & Structured Outputs",
      "question": "Explain the difference between a system prompt, developer prompt, and user prompt in modern chat APIs.",
      "difficulty": 2
    },
    {
      "id": 353,
      "section": "Section 9 — Prompt Engineering & Structured Outputs",
      "question": "How do you design prompts to reduce hallucination on factual questions?",
      "difficulty": 3
    },
    {
      "id": 354,
      "section": "Section 9 — Prompt Engineering & Structured Outputs",
      "question": "What is prompt injection, and how does it differ from jailbreaking?",
      "difficulty": 2
    },
    {
      "id": 355,
      "section": "Section 9 — Prompt Engineering & Structured Outputs",
      "question": "Explain few-shot example selection strategies (similarity-based retrieval of examples).",
      "difficulty": 3
    },
    {
      "id": 356,
      "section": "Section 9 — Prompt Engineering & Structured Outputs",
      "question": "How do you version and test prompts systematically as they evolve?",
      "difficulty": 2
    },
    {
      "id": 357,
      "section": "Section 9 — Prompt Engineering & Structured Outputs",
      "question": "What is prompt compression, and why does it matter for cost at scale?",
      "difficulty": 2
    },
    {
      "id": 358,
      "section": "Section 9 — Prompt Engineering & Structured Outputs",
      "question": "Explain structured output generation via JSON mode vs. function/tool calling.",
      "difficulty": 3
    },
    {
      "id": 359,
      "section": "Section 9 — Prompt Engineering & Structured Outputs",
      "question": "What is the role of a schema (e.g., Pydantic/JSON Schema) in constraining LLM output?",
      "difficulty": 3
    },
    {
      "id": 360,
      "section": "Section 9 — Prompt Engineering & Structured Outputs",
      "question": "Explain grammar-constrained decoding for guaranteed structured output.",
      "difficulty": 3
    },
    {
      "id": 361,
      "section": "Section 9 — Prompt Engineering & Structured Outputs",
      "question": "How do you handle malformed JSON output from an LLM in production?",
      "difficulty": 3
    },
    {
      "id": 362,
      "section": "Section 9 — Prompt Engineering & Structured Outputs",
      "question": "What is function calling, and how does the model decide which function to call?",
      "difficulty": 2
    },
    {
      "id": 363,
      "section": "Section 9 — Prompt Engineering & Structured Outputs",
      "question": "Explain multi-tool selection — how does an LLM choose among many available tools?",
      "difficulty": 3
    },
    {
      "id": 364,
      "section": "Section 9 — Prompt Engineering & Structured Outputs",
      "question": "How do you design tool descriptions to minimize incorrect tool selection?",
      "difficulty": 2
    },
    {
      "id": 365,
      "section": "Section 9 — Prompt Engineering & Structured Outputs",
      "question": "What is retrieval-augmented prompting, and how does it differ from full RAG pipelines?",
      "difficulty": 3
    },
    {
      "id": 366,
      "section": "Section 9 — Prompt Engineering & Structured Outputs",
      "question": "Explain the tradeoffs of long, detailed system prompts vs. short ones with examples.",
      "difficulty": 2
    },
    {
      "id": 367,
      "section": "Section 9 — Prompt Engineering & Structured Outputs",
      "question": "How do you test prompts for robustness across paraphrased inputs?",
      "difficulty": 3
    },
    {
      "id": 368,
      "section": "Section 9 — Prompt Engineering & Structured Outputs",
      "question": "What is prompt leaking, and how do you defend against it?",
      "difficulty": 2
    },
    {
      "id": 369,
      "section": "Section 9 — Prompt Engineering & Structured Outputs",
      "question": "Explain output parsing strategies when a model doesn't reliably follow a schema.",
      "difficulty": 2
    },
    {
      "id": 370,
      "section": "Section 9 — Prompt Engineering & Structured Outputs",
      "question": "How would you A/B test two prompt variants in production?",
      "difficulty": 3
    },
    {
      "id": 371,
      "section": "Section 9 — Prompt Engineering & Structured Outputs",
      "question": "What is meta-prompting (using an LLM to generate/improve prompts)?",
      "difficulty": 3
    },
    {
      "id": 372,
      "section": "Section 9 — Prompt Engineering & Structured Outputs",
      "question": "Explain the risk of prompt overfitting to a narrow eval set.",
      "difficulty": 2
    },
    {
      "id": 373,
      "section": "Section 9 — Prompt Engineering & Structured Outputs",
      "question": "How do you handle multilingual prompting consistently across languages?",
      "difficulty": 3
    },
    {
      "id": 374,
      "section": "Section 9 — Prompt Engineering & Structured Outputs",
      "question": "What role does few-shot example ordering play in output quality?",
      "difficulty": 3
    },
    {
      "id": 375,
      "section": "Section 9 — Prompt Engineering & Structured Outputs",
      "question": "Explain the difference between instructing a model \"what to do\" vs. \"what not to do,\" and which tends to work better.",
      "difficulty": 2
    },
    {
      "id": 376,
      "section": "Section 10 — RAG & Retrieval",
      "question": "Design a RAG system answering questions over 50M internal documents end to end.",
      "difficulty": 3
    },
    {
      "id": 377,
      "section": "Section 10 — RAG & Retrieval",
      "question": "Explain the RAG pipeline: chunking, embedding, indexing, retrieval, re-ranking, generation.",
      "difficulty": 3
    },
    {
      "id": 378,
      "section": "Section 10 — RAG & Retrieval",
      "question": "What chunking strategies exist (fixed-size, semantic, recursive, sentence-window), and how do you choose?",
      "difficulty": 3
    },
    {
      "id": 379,
      "section": "Section 10 — RAG & Retrieval",
      "question": "Explain the tradeoff between chunk size and retrieval precision/recall.",
      "difficulty": 3
    },
    {
      "id": 380,
      "section": "Section 10 — RAG & Retrieval",
      "question": "What is chunk overlap, and why is it used?",
      "difficulty": 3
    },
    {
      "id": 381,
      "section": "Section 10 — RAG & Retrieval",
      "question": "Explain dense retrieval vs. sparse retrieval (BM25) vs. hybrid retrieval.",
      "difficulty": 3
    },
    {
      "id": 382,
      "section": "Section 10 — RAG & Retrieval",
      "question": "What is re-ranking, and why is a two-stage retrieve-then-rerank pipeline often better than retrieval alone?",
      "difficulty": 2
    },
    {
      "id": 383,
      "section": "Section 10 — RAG & Retrieval",
      "question": "Explain cross-encoder vs. bi-encoder re-rankers — tradeoffs?",
      "difficulty": 2
    },
    {
      "id": 384,
      "section": "Section 10 — RAG & Retrieval",
      "question": "What is query expansion/rewriting, and how does it improve retrieval?",
      "difficulty": 3
    },
    {
      "id": 385,
      "section": "Section 10 — RAG & Retrieval",
      "question": "Explain HyDE (Hypothetical Document Embeddings) as a retrieval technique.",
      "difficulty": 3
    },
    {
      "id": 386,
      "section": "Section 10 — RAG & Retrieval",
      "question": "What is multi-hop retrieval, and when is it necessary?",
      "difficulty": 2
    },
    {
      "id": 387,
      "section": "Section 10 — RAG & Retrieval",
      "question": "Explain how you'd handle document freshness/staleness in a RAG index.",
      "difficulty": 3
    },
    {
      "id": 388,
      "section": "Section 10 — RAG & Retrieval",
      "question": "What strategies exist for handling structured data (tables) inside a RAG pipeline?",
      "difficulty": 3
    },
    {
      "id": 389,
      "section": "Section 10 — RAG & Retrieval",
      "question": "Explain parent-document retrieval (small-to-big chunking).",
      "difficulty": 2
    },
    {
      "id": 390,
      "section": "Section 10 — RAG & Retrieval",
      "question": "What is contextual compression in RAG, and how does it reduce prompt size?",
      "difficulty": 3
    },
    {
      "id": 391,
      "section": "Section 10 — RAG & Retrieval",
      "question": "Explain how you'd evaluate a RAG system's retrieval quality separately from generation quality.",
      "difficulty": 3
    },
    {
      "id": 392,
      "section": "Section 10 — RAG & Retrieval",
      "question": "What metrics measure retrieval quality (recall@k, MRR, NDCG)?",
      "difficulty": 2
    },
    {
      "id": 393,
      "section": "Section 10 — RAG & Retrieval",
      "question": "Explain groundedness/faithfulness evaluation for RAG-generated answers.",
      "difficulty": 3
    },
    {
      "id": 394,
      "section": "Section 10 — RAG & Retrieval",
      "question": "What is citation/attribution in RAG output, and how do you enforce it?",
      "difficulty": 3
    },
    {
      "id": 395,
      "section": "Section 10 — RAG & Retrieval",
      "question": "Explain how you'd design RAG for multi-tenant data isolation (per-customer document access).",
      "difficulty": 3
    },
    {
      "id": 396,
      "section": "Section 10 — RAG & Retrieval",
      "question": "What's your approach to access control (row-level security) inside a shared vector index?",
      "difficulty": 3
    },
    {
      "id": 397,
      "section": "Section 10 — RAG & Retrieval",
      "question": "Explain agentic RAG — where the model decides when and what to retrieve.",
      "difficulty": 3
    },
    {
      "id": 398,
      "section": "Section 10 — RAG & Retrieval",
      "question": "What is GraphRAG, and when does a knowledge-graph-augmented approach outperform vector-only RAG?",
      "difficulty": 2
    },
    {
      "id": 399,
      "section": "Section 10 — RAG & Retrieval",
      "question": "Explain how you'd combine RAG with fine-tuning for a domain-specific assistant.",
      "difficulty": 3
    },
    {
      "id": 400,
      "section": "Section 10 — RAG & Retrieval",
      "question": "What is the \"lost in the middle\" problem for long-context LLMs, and how does RAG mitigate or worsen it?",
      "difficulty": 2
    },
    {
      "id": 401,
      "section": "Section 10 — RAG & Retrieval",
      "question": "Explain how you'd design RAG evaluation with no ground-truth labeled Q&A pairs.",
      "difficulty": 2
    },
    {
      "id": 402,
      "section": "Section 10 — RAG & Retrieval",
      "question": "What is self-RAG / corrective RAG, and how does it improve reliability?",
      "difficulty": 3
    },
    {
      "id": 403,
      "section": "Section 10 — RAG & Retrieval",
      "question": "Explain how you'd handle conflicting information across retrieved documents.",
      "difficulty": 3
    },
    {
      "id": 404,
      "section": "Section 10 — RAG & Retrieval",
      "question": "What caching strategies exist for RAG (embedding cache, retrieval cache, semantic cache)?",
      "difficulty": 2
    },
    {
      "id": 405,
      "section": "Section 10 — RAG & Retrieval",
      "question": "Explain how you'd scale a RAG index from 1M to 1B documents.",
      "difficulty": 2
    },
    {
      "id": 406,
      "section": "Section 10 — RAG & Retrieval",
      "question": "What's your approach to incremental indexing vs. full reindexing when documents change?",
      "difficulty": 3
    },
    {
      "id": 407,
      "section": "Section 10 — RAG & Retrieval",
      "question": "Explain multimodal RAG — retrieving over images, tables, and text together.",
      "difficulty": 2
    },
    {
      "id": 408,
      "section": "Section 10 — RAG & Retrieval",
      "question": "How do you handle PII and sensitive data inside a RAG corpus?",
      "difficulty": 2
    },
    {
      "id": 409,
      "section": "Section 10 — RAG & Retrieval",
      "question": "Explain how document metadata (filters, tags) is used to narrow retrieval before the vector search.",
      "difficulty": 2
    },
    {
      "id": 410,
      "section": "Section 10 — RAG & Retrieval",
      "question": "What is the role of embedding model choice, and how do you evaluate/select one?",
      "difficulty": 2
    },
    {
      "id": 411,
      "section": "Section 10 — RAG & Retrieval",
      "question": "Explain how you'd fine-tune an embedding model for a domain-specific retrieval task.",
      "difficulty": 2
    },
    {
      "id": 412,
      "section": "Section 10 — RAG & Retrieval",
      "question": "What is negative mining, and how is it used to train better retrieval/embedding models?",
      "difficulty": 2
    },
    {
      "id": 413,
      "section": "Section 10 — RAG & Retrieval",
      "question": "Explain how you'd detect and handle retrieval failure (no relevant document found) gracefully.",
      "difficulty": 3
    },
    {
      "id": 414,
      "section": "Section 10 — RAG & Retrieval",
      "question": "What's the tradeoff between retrieving more chunks (higher recall) vs. fewer (lower noise, lower cost)?",
      "difficulty": 2
    },
    {
      "id": 415,
      "section": "Section 10 — RAG & Retrieval",
      "question": "Explain how you'd design a RAG system to cite exact source passages, not just document titles.",
      "difficulty": 3
    },
    {
      "id": 416,
      "section": "Section 10 — RAG & Retrieval",
      "question": "What is late chunking / late interaction (ColBERT-style), and how does it differ from standard dense retrieval?",
      "difficulty": 3
    },
    {
      "id": 417,
      "section": "Section 10 — RAG & Retrieval",
      "question": "Explain how summarization of retrieved chunks before generation can help or hurt answer quality.",
      "difficulty": 2
    },
    {
      "id": 418,
      "section": "Section 10 — RAG & Retrieval",
      "question": "What's your strategy for RAG over code repositories specifically (as opposed to prose documents)?",
      "difficulty": 3
    },
    {
      "id": 419,
      "section": "Section 10 — RAG & Retrieval",
      "question": "Explain how you'd design retrieval for conversational (multi-turn) RAG where context depends on prior turns.",
      "difficulty": 3
    },
    {
      "id": 420,
      "section": "Section 10 — RAG & Retrieval",
      "question": "How would you diagnose a RAG system that retrieves relevant chunks but still generates wrong answers?",
      "difficulty": 3
    },
    {
      "id": 421,
      "section": "Section 11 — Vector Databases & Embeddings",
      "question": "Explain how vector databases (Pinecone, Weaviate, Milvus, pgvector, FAISS) differ architecturally.",
      "difficulty": 2
    },
    {
      "id": 422,
      "section": "Section 11 — Vector Databases & Embeddings",
      "question": "What is approximate nearest neighbor (ANN) search, and why not use exact kNN at scale?",
      "difficulty": 2
    },
    {
      "id": 423,
      "section": "Section 11 — Vector Databases & Embeddings",
      "question": "Explain HNSW (Hierarchical Navigable Small World) indexing at a conceptual level.",
      "difficulty": 2
    },
    {
      "id": 424,
      "section": "Section 11 — Vector Databases & Embeddings",
      "question": "Compare IVF (Inverted File Index) and HNSW — tradeoffs in build time, query speed, recall.",
      "difficulty": 2
    },
    {
      "id": 425,
      "section": "Section 11 — Vector Databases & Embeddings",
      "question": "What is product quantization, and how does it reduce vector storage cost?",
      "difficulty": 3
    },
    {
      "id": 426,
      "section": "Section 11 — Vector Databases & Embeddings",
      "question": "Explain the recall-latency tradeoff in ANN search and how you'd tune it.",
      "difficulty": 2
    },
    {
      "id": 427,
      "section": "Section 11 — Vector Databases & Embeddings",
      "question": "What is hybrid search (dense + sparse), and how do you combine scores (e.g., reciprocal rank fusion)?",
      "difficulty": 2
    },
    {
      "id": 428,
      "section": "Section 11 — Vector Databases & Embeddings",
      "question": "Explain metadata filtering in vector search and its performance implications.",
      "difficulty": 3
    },
    {
      "id": 429,
      "section": "Section 11 — Vector Databases & Embeddings",
      "question": "What is embedding dimensionality's tradeoff — higher dims vs. storage/compute cost?",
      "difficulty": 2
    },
    {
      "id": 430,
      "section": "Section 11 — Vector Databases & Embeddings",
      "question": "Explain how you'd choose between a managed vector DB and a self-hosted one (pgvector on Postgres).",
      "difficulty": 2
    },
    {
      "id": 431,
      "section": "Section 11 — Vector Databases & Embeddings",
      "question": "What is index rebuild cost, and how do you handle it for a continuously updated corpus?",
      "difficulty": 2
    },
    {
      "id": 432,
      "section": "Section 11 — Vector Databases & Embeddings",
      "question": "Explain sharding strategies for a vector database at billion-scale.",
      "difficulty": 3
    },
    {
      "id": 433,
      "section": "Section 11 — Vector Databases & Embeddings",
      "question": "What is embedding drift, and how would you detect that your embedding model needs updating?",
      "difficulty": 2
    },
    {
      "id": 434,
      "section": "Section 11 — Vector Databases & Embeddings",
      "question": "Explain multi-vector representations (e.g., ColBERT) vs. single-vector embeddings.",
      "difficulty": 3
    },
    {
      "id": 435,
      "section": "Section 11 — Vector Databases & Embeddings",
      "question": "What's your approach to embedding versioning when you update the embedding model?",
      "difficulty": 3
    },
    {
      "id": 436,
      "section": "Section 11 — Vector Databases & Embeddings",
      "question": "Explain how you'd benchmark different vector databases for a specific workload.",
      "difficulty": 3
    },
    {
      "id": 437,
      "section": "Section 11 — Vector Databases & Embeddings",
      "question": "What is quantization within vector DBs (scalar/binary quantization), and its accuracy tradeoff?",
      "difficulty": 3
    },
    {
      "id": 438,
      "section": "Section 11 — Vector Databases & Embeddings",
      "question": "Explain how you'd handle multi-tenancy and namespace isolation in a shared vector DB.",
      "difficulty": 3
    },
    {
      "id": 439,
      "section": "Section 11 — Vector Databases & Embeddings",
      "question": "What is the cost model for a vector DB at scale (storage, compute, query throughput)?",
      "difficulty": 2
    },
    {
      "id": 440,
      "section": "Section 11 — Vector Databases & Embeddings",
      "question": "Explain how embeddings for text, image, and code differ, and whether they can share a vector space.",
      "difficulty": 3
    },
    {
      "id": 441,
      "section": "Section 11 — Vector Databases & Embeddings",
      "question": "What is a vector index's \"recall floor,\" and how would you set a minimum acceptable recall threshold before shipping a retrieval feature to production?",
      "difficulty": 2
    },
    {
      "id": 442,
      "section": "Section 11 — Vector Databases & Embeddings",
      "question": "Explain how you'd migrate a production vector index from one embedding model to another with zero retrieval downtime.",
      "difficulty": 2
    },
    {
      "id": 443,
      "section": "Section 11 — Vector Databases & Embeddings",
      "question": "What is the tradeoff between storing full-precision vectors versus binary/scalar-quantized vectors for a cost-sensitive, large-scale deployment?",
      "difficulty": 2
    },
    {
      "id": 444,
      "section": "Section 11 — Vector Databases & Embeddings",
      "question": "Explain how vector database choice interacts with your broader data platform — when does it make sense to add vector search directly to an existing operational database (e.g., Postgres/pgvector) versus standing up a dedicated vector database?",
      "difficulty": 2
    },
    {
      "id": 445,
      "section": "Section 11 — Vector Databases & Embeddings",
      "question": "What is the role of a vector database's write consistency model, and why does eventual consistency matter for a RAG system with frequently updated documents?",
      "difficulty": 2
    },
    {
      "id": 446,
      "section": "Section 11 — Vector Databases & Embeddings",
      "question": "Explain how you'd benchmark vector search cost (not just latency/recall) across candidate providers at your actual production scale.",
      "difficulty": 2
    },
    {
      "id": 447,
      "section": "Section 11 — Vector Databases & Embeddings",
      "question": "What is a \"hot\" versus \"cold\" partition strategy for a vector index serving both frequently-queried recent documents and a long tail of rarely-queried historical ones?",
      "difficulty": 2
    },
    {
      "id": 448,
      "section": "Section 11 — Vector Databases & Embeddings",
      "question": "Explain how filtered vector search (metadata pre-filtering) performance degrades when filters are highly selective, and how index design should account for it.",
      "difficulty": 2
    },
    {
      "id": 449,
      "section": "Section 11 — Vector Databases & Embeddings",
      "question": "What operational monitoring would you put on a production vector database beyond query latency — index size growth, memory pressure, and recall drift over time?",
      "difficulty": 2
    },
    {
      "id": 450,
      "section": "Section 11 — Vector Databases & Embeddings",
      "question": "Explain cross-lingual embeddings and their use in multilingual retrieval.",
      "difficulty": 3
    },
    {
      "id": 451,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "Explain the ReAct pattern (reason, act, observe) for building tool-using agents.",
      "difficulty": 3
    },
    {
      "id": 452,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "What is an agent's \"scratchpad\" or working memory, and how is it maintained across steps?",
      "difficulty": 2
    },
    {
      "id": 453,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "Explain planning vs. execution separation in agent architectures.",
      "difficulty": 3
    },
    {
      "id": 454,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "What is a plan-and-execute agent, and how does it differ from a single-loop ReAct agent?",
      "difficulty": 3
    },
    {
      "id": 455,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "Explain how you'd design multi-agent orchestration for a complex workflow (planning, state, tool use, cost control).",
      "difficulty": 2
    },
    {
      "id": 456,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "What is the orchestrator-worker pattern in multi-agent systems?",
      "difficulty": 3
    },
    {
      "id": 457,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "Explain how agents communicate state to each other (shared memory, message passing, blackboard pattern).",
      "difficulty": 3
    },
    {
      "id": 458,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "What is tool calling, and how does an agent decide which tool to invoke and with what arguments?",
      "difficulty": 2
    },
    {
      "id": 459,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "Explain how you'd design error handling and retries for a tool call that fails mid-task.",
      "difficulty": 3
    },
    {
      "id": 460,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "What's your approach to bounding an agent's action space to prevent runaway or destructive actions?",
      "difficulty": 2
    },
    {
      "id": 461,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "Explain how you'd design a human-in-the-loop checkpoint for high-risk agent actions.",
      "difficulty": 3
    },
    {
      "id": 462,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "What is agent memory — short-term (context window) vs. long-term (persistent store)?",
      "difficulty": 3
    },
    {
      "id": 463,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "Explain how you'd implement long-term memory for an agent across sessions.",
      "difficulty": 3
    },
    {
      "id": 464,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "What is the \"lost context\" problem in long-running agent loops, and how do you mitigate it?",
      "difficulty": 2
    },
    {
      "id": 465,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "Explain how you'd design cost controls (token budgets, step limits) for an autonomous agent.",
      "difficulty": 2
    },
    {
      "id": 466,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "What is a supervisor/critic agent pattern, and when does it improve reliability?",
      "difficulty": 2
    },
    {
      "id": 467,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "Explain how you'd evaluate a multi-agent system's end-to-end task success rate.",
      "difficulty": 2
    },
    {
      "id": 468,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "What is agent looping/getting stuck, and how do you detect and break out of it?",
      "difficulty": 3
    },
    {
      "id": 469,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "Explain how you'd design an agent to gracefully hand off to a human when it's uncertain.",
      "difficulty": 2
    },
    {
      "id": 470,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "What's the difference between a single powerful agent and a swarm of specialized agents?",
      "difficulty": 2
    },
    {
      "id": 471,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "Explain how you'd design state persistence for a long-running (hours/days) agent workflow.",
      "difficulty": 2
    },
    {
      "id": 472,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "What is LangGraph's graph-based approach to agent orchestration, and when would you choose it over a simple loop?",
      "difficulty": 3
    },
    {
      "id": 473,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "Explain how you'd design an agent's tool interface to minimize hallucinated tool calls.",
      "difficulty": 3
    },
    {
      "id": 474,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "What is the role of a verifier/validator step after an agent produces output?",
      "difficulty": 3
    },
    {
      "id": 475,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "Explain how you'd design agent-to-agent negotiation or delegation in a multi-agent workflow.",
      "difficulty": 3
    },
    {
      "id": 476,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "What tradeoffs exist between giving an agent more tools vs. fewer, more composable ones?",
      "difficulty": 3
    },
    {
      "id": 477,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "Explain how you'd secure an agent that has access to sensitive systems (databases, payment APIs).",
      "difficulty": 2
    },
    {
      "id": 478,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "What is prompt injection risk specifically for tool-using agents (e.g., malicious content in a retrieved doc)?",
      "difficulty": 2
    },
    {
      "id": 479,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "Explain sandboxing strategies for code-executing agents.",
      "difficulty": 3
    },
    {
      "id": 480,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "What's your approach to testing an agent's behavior against adversarial or edge-case inputs?",
      "difficulty": 3
    },
    {
      "id": 481,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "Explain how you'd design rollback/undo capability for agent actions that modify external state.",
      "difficulty": 2
    },
    {
      "id": 482,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "What is a \"critic\" or self-reflection loop, and how does it improve agent output quality?",
      "difficulty": 3
    },
    {
      "id": 483,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "Explain how you'd design an agent that must complete a task within a hard deadline/budget.",
      "difficulty": 3
    },
    {
      "id": 484,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "What's the difference between deterministic workflow automation (e.g., n8n) and LLM-driven agentic automation?",
      "difficulty": 2
    },
    {
      "id": 485,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "Explain how you'd choose between a rules-based system, a workflow engine, and an autonomous agent for a given task.",
      "difficulty": 3
    },
    {
      "id": 486,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "What is the role of observability (tracing, logging) in debugging multi-agent systems?",
      "difficulty": 3
    },
    {
      "id": 487,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "Explain how you'd design an agent evaluation harness that simulates realistic multi-turn user interactions.",
      "difficulty": 3
    },
    {
      "id": 488,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "What is context window management across a long multi-agent conversation, and how do you summarize/prune it?",
      "difficulty": 3
    },
    {
      "id": 489,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "Explain how you'd prevent two agents from entering an infinite back-and-forth loop.",
      "difficulty": 2
    },
    {
      "id": 490,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "What is the \"single responsibility\" principle applied to agent design, and why does it improve reliability?",
      "difficulty": 3
    },
    {
      "id": 491,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "Explain how you'd design cost attribution when multiple agents/tools contribute to a single user request.",
      "difficulty": 2
    },
    {
      "id": 492,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "What's your approach to versioning agent behavior as prompts, tools, and models change over time?",
      "difficulty": 3
    },
    {
      "id": 493,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "Explain how you'd design an agent for a regulated domain (e.g., finance, healthcare) with audit requirements.",
      "difficulty": 3
    },
    {
      "id": 494,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "What is the risk of agents taking real-world actions (bookings, payments, emails) without sufficient guardrails?",
      "difficulty": 3
    },
    {
      "id": 495,
      "section": "Section 12 — Agentic AI & Multi-Agent Systems",
      "question": "Explain how you'd design a fallback path when an agent's confidence is low.",
      "difficulty": 2
    },
    {
      "id": 496,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "Design a customer-support chatbot backed by an LLM with tool-calling and escalation to a human.",
      "difficulty": 3
    },
    {
      "id": 497,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "Design a code-review agent that integrates with a CI pipeline.",
      "difficulty": 3
    },
    {
      "id": 498,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "Design a document summarization pipeline at scale — cost, latency, accuracy tradeoffs.",
      "difficulty": 3
    },
    {
      "id": 499,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "Design a semantic search/embedding service — index choice, recall vs. latency, hybrid search.",
      "difficulty": 2
    },
    {
      "id": 500,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "Design real-time streaming chat — token streaming, session memory, backpressure.",
      "difficulty": 3
    },
    {
      "id": 501,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "Design a multimodal (vision-language) serving pipeline — image token budget, latency.",
      "difficulty": 2
    },
    {
      "id": 502,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "How do you reduce LLM serving cost without hurting quality (routing, cascades, caching, quantization)?",
      "difficulty": 2
    },
    {
      "id": 503,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "Design a model-routing system across multiple LLMs of different sizes/costs for one product.",
      "difficulty": 3
    },
    {
      "id": 504,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "How would you architect a system that falls back to a smaller/cheaper model under load?",
      "difficulty": 2
    },
    {
      "id": 505,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "Design an LLM-powered email-drafting assistant integrated into an existing product.",
      "difficulty": 2
    },
    {
      "id": 506,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "Design a system for LLM-based data extraction from unstructured documents (invoices, contracts) at scale.",
      "difficulty": 3
    },
    {
      "id": 507,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "How would you design a translation service backed by an LLM with terminology consistency requirements?",
      "difficulty": 2
    },
    {
      "id": 508,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "Design a voice assistant pipeline (ASR → LLM → TTS) with end-to-end latency targets.",
      "difficulty": 3
    },
    {
      "id": 509,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "Design an internal \"ask your company's data\" assistant spanning multiple data sources (docs, tickets, DBs).",
      "difficulty": 2
    },
    {
      "id": 510,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "How would you architect a system that must support both synchronous chat and long-running async jobs?",
      "difficulty": 2
    },
    {
      "id": 511,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "Design a content-moderation pipeline that combines classic classifiers with an LLM judge.",
      "difficulty": 3
    },
    {
      "id": 512,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "Design a personalization system that blends collaborative filtering with LLM-based re-ranking.",
      "difficulty": 3
    },
    {
      "id": 513,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "How would you design a system to generate and validate SQL from natural language safely?",
      "difficulty": 3
    },
    {
      "id": 514,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "Design an LLM-powered search-ranking re-ranker layered on top of an existing search engine.",
      "difficulty": 2
    },
    {
      "id": 515,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "How would you design a system for LLM-assisted code generation with test-driven validation before merge?",
      "difficulty": 3
    },
    {
      "id": 516,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "Design a knowledge-base-updating pipeline where an LLM proposes edits that a human approves.",
      "difficulty": 3
    },
    {
      "id": 517,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "How would you architect an LLM gateway/proxy layer for a company with 50+ internal AI consumers?",
      "difficulty": 3
    },
    {
      "id": 518,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "Design rate limiting and quota management for a multi-tenant LLM API platform.",
      "difficulty": 3
    },
    {
      "id": 519,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "How would you design request routing to balance latency, cost, and quality across model tiers?",
      "difficulty": 3
    },
    {
      "id": 520,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "Design a caching layer for LLM responses (exact-match and semantic caching) and its invalidation strategy.",
      "difficulty": 3
    },
    {
      "id": 521,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "How would you design an LLM-based fraud-narrative summarizer for investigators, with strict factuality requirements?",
      "difficulty": 3
    },
    {
      "id": 522,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "Design a system for automatically generating release notes from commit history using an LLM.",
      "difficulty": 2
    },
    {
      "id": 523,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "How would you design a multilingual customer support system with consistent quality across languages?",
      "difficulty": 3
    },
    {
      "id": 524,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "Design an LLM-based resume-screening system, accounting for fairness and legal risk.",
      "difficulty": 2
    },
    {
      "id": 525,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "How would you design an LLM system for meeting-transcript summarization with speaker attribution?",
      "difficulty": 2
    },
    {
      "id": 526,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "Design an architecture for A/B testing different LLM providers on live traffic safely.",
      "difficulty": 3
    },
    {
      "id": 527,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "How would you design graceful degradation when your primary LLM provider has an outage?",
      "difficulty": 2
    },
    {
      "id": 528,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "Design an LLM-based anomaly-explanation system for an existing monitoring/alerting platform.",
      "difficulty": 2
    },
    {
      "id": 529,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "How would you architect an LLM system that must comply with strict data-residency requirements (EU-only data)?",
      "difficulty": 2
    },
    {
      "id": 530,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "Design a \"co-pilot\" feature embedded inside an existing SaaS product — how do you scope its permissions?",
      "difficulty": 2
    },
    {
      "id": 531,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "How would you design cost forecasting/budgeting for an LLM feature before it launches?",
      "difficulty": 2
    },
    {
      "id": 532,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "Design a system for continuous prompt/model regression testing tied to CI/CD.",
      "difficulty": 2
    },
    {
      "id": 533,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "How would you architect logging and observability for an LLM product (traces, token usage, latency, quality)?",
      "difficulty": 2
    },
    {
      "id": 534,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "Design a system to detect and redact PII before it reaches an external LLM provider.",
      "difficulty": 2
    },
    {
      "id": 535,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "How would you design an LLM feature to work offline/on-device for a mobile app with connectivity gaps?",
      "difficulty": 2
    },
    {
      "id": 536,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "Design a system for generating structured reports (e.g., financial summaries) from LLM output with human sign-off.",
      "difficulty": 2
    },
    {
      "id": 537,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "How would you design version pinning/rollback for an LLM-powered feature when the underlying model updates?",
      "difficulty": 3
    },
    {
      "id": 538,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "Design a system that lets non-technical users build and deploy their own prompts/agents safely (a low-code AI platform).",
      "difficulty": 2
    },
    {
      "id": 539,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "How would you design a shared \"prompt library\" and versioning system across many product teams?",
      "difficulty": 2
    },
    {
      "id": 540,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "Design an architecture where multiple LLM calls in a pipeline must stay under a strict end-to-end latency SLA.",
      "difficulty": 3
    },
    {
      "id": 541,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "How would you design a system for detecting when an LLM-based feature's quality degrades in production silently?",
      "difficulty": 3
    },
    {
      "id": 542,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "Design a \"playground\" internal tool for engineers to test prompts against multiple models before shipping.",
      "difficulty": 3
    },
    {
      "id": 543,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "How would you design a chat product's conversation-history storage for both product features and compliance/audit needs?",
      "difficulty": 2
    },
    {
      "id": 544,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "Design a system for summarizing legal contracts with clause-level citations back to source text.",
      "difficulty": 2
    },
    {
      "id": 545,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "How would you architect an LLM feature for a high-throughput, low-latency ad-serving context?",
      "difficulty": 2
    },
    {
      "id": 546,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "Design a system to synthesize training data using an LLM for a smaller downstream fine-tuned model.",
      "difficulty": 3
    },
    {
      "id": 547,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "How would you design cost-aware prompt truncation when a conversation exceeds the context window?",
      "difficulty": 2
    },
    {
      "id": 548,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "Design a system that routes user queries to either a deterministic FAQ system or an LLM based on confidence.",
      "difficulty": 2
    },
    {
      "id": 549,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "How would you architect a review/approval workflow for AI-generated marketing content before publishing?",
      "difficulty": 3
    },
    {
      "id": 550,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "Design an LLM-powered onboarding assistant that must stay strictly within a defined product scope.",
      "difficulty": 3
    },
    {
      "id": 551,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "How would you design a system that lets you swap the underlying LLM provider with minimal code changes?",
      "difficulty": 3
    },
    {
      "id": 552,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "Design an architecture for handling long-document Q&A (100+ page PDFs) with citation accuracy.",
      "difficulty": 2
    },
    {
      "id": 553,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "How would you design a \"confidence score\" surfaced to end users for LLM-generated answers?",
      "difficulty": 2
    },
    {
      "id": 554,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "Design a system to detect and prevent prompt injection from user-uploaded documents in a RAG pipeline.",
      "difficulty": 2
    },
    {
      "id": 555,
      "section": "Section 13 — LLM System Design / GenAI Architecture",
      "question": "How would you architect disaster recovery for a mission-critical LLM-powered production system?",
      "difficulty": 2
    },
    {
      "id": 556,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design a recommendation system for an e-commerce platform end to end.",
      "difficulty": 2
    },
    {
      "id": 557,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design a fraud/anomaly-detection system requiring low latency and adaptation to concept drift.",
      "difficulty": 2
    },
    {
      "id": 558,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design a search-ranking system combining classic ML and LLM re-ranking.",
      "difficulty": 2
    },
    {
      "id": 559,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design an ML feature store used by multiple teams — how do you ensure training/serving consistency?",
      "difficulty": 2
    },
    {
      "id": 560,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design a system for real-time (online) inference vs. batch (offline) inference — when do you pick each?",
      "difficulty": 2
    },
    {
      "id": 561,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design a model-monitoring system: data drift, prediction drift, and performance decay detection.",
      "difficulty": 2
    },
    {
      "id": 562,
      "section": "Section 14 — Classic ML System Design",
      "question": "How would you design safe A/B testing of model versions in production?",
      "difficulty": 2
    },
    {
      "id": 563,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design a credit-risk scoring system with regulatory explainability requirements.",
      "difficulty": 2
    },
    {
      "id": 564,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design a dynamic pricing system that must react to real-time demand signals.",
      "difficulty": 2
    },
    {
      "id": 565,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design a churn-prediction system feeding into an automated retention-campaign trigger.",
      "difficulty": 2
    },
    {
      "id": 566,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design an ad click-through-rate (CTR) prediction system at scale.",
      "difficulty": 2
    },
    {
      "id": 567,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design a search-query autocomplete system with sub-100ms latency.",
      "difficulty": 2
    },
    {
      "id": 568,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design an image-based product-search system (visual search) for e-commerce.",
      "difficulty": 2
    },
    {
      "id": 569,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design a spam/abuse-detection system for user-generated content at platform scale.",
      "difficulty": 2
    },
    {
      "id": 570,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design a demand-forecasting system for retail inventory planning.",
      "difficulty": 2
    },
    {
      "id": 571,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design a ride-sharing ETA-prediction system.",
      "difficulty": 2
    },
    {
      "id": 572,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design a video-recommendation system balancing engagement and content diversity.",
      "difficulty": 2
    },
    {
      "id": 573,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design a system detecting duplicate/near-duplicate content at scale.",
      "difficulty": 2
    },
    {
      "id": 574,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design a real-time bidding system for programmatic advertising.",
      "difficulty": 2
    },
    {
      "id": 575,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design a credit-card fraud system that must decide within 100ms per transaction.",
      "difficulty": 2
    },
    {
      "id": 576,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design a system for detecting fake reviews or fake accounts.",
      "difficulty": 2
    },
    {
      "id": 577,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design a next-best-action recommendation system for a sales team.",
      "difficulty": 2
    },
    {
      "id": 578,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design a system for predictive maintenance using sensor/IoT data.",
      "difficulty": 2
    },
    {
      "id": 579,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design a system that ranks support tickets by urgency for a customer-service team.",
      "difficulty": 2
    },
    {
      "id": 580,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design an ML system for matching job candidates to postings, with fairness constraints.",
      "difficulty": 2
    },
    {
      "id": 581,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design a system for detecting network intrusions/anomalies in real time.",
      "difficulty": 2
    },
    {
      "id": 582,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design a personalized email send-time-optimization system.",
      "difficulty": 2
    },
    {
      "id": 583,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design a system that predicts and prevents cart abandonment in real time.",
      "difficulty": 2
    },
    {
      "id": 584,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design an inventory-allocation optimization system across warehouses.",
      "difficulty": 2
    },
    {
      "id": 585,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design a system for real-time language detection and routing in a global support platform.",
      "difficulty": 2
    },
    {
      "id": 586,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design an ML pipeline for predicting equipment failure from time-series sensor data.",
      "difficulty": 2
    },
    {
      "id": 587,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design a system to detect coordinated inauthentic behavior (bot networks) on a social platform.",
      "difficulty": 2
    },
    {
      "id": 588,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design a system for personalized notification-frequency capping to reduce churn from over-notification.",
      "difficulty": 2
    },
    {
      "id": 589,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design an ML system for insurance claim triage and fraud flagging.",
      "difficulty": 2
    },
    {
      "id": 590,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design a system that predicts server capacity needs for autoscaling.",
      "difficulty": 2
    },
    {
      "id": 591,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design a system for real-time bid optimization in a marketing budget-allocation tool.",
      "difficulty": 2
    },
    {
      "id": 592,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design a system for detecting toxic/harassing content in live chat with low false-positive rate.",
      "difficulty": 2
    },
    {
      "id": 593,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design a system to personalize search-result ranking per user without leaking data across users.",
      "difficulty": 2
    },
    {
      "id": 594,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design a system for automatic tagging/categorization of a growing content catalog.",
      "difficulty": 2
    },
    {
      "id": 595,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design a real-time recommendation system with a strict \"session must feel fresh\" requirement.",
      "difficulty": 2
    },
    {
      "id": 596,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design a system for detecting label-quality issues in a crowd-sourced annotation pipeline.",
      "difficulty": 2
    },
    {
      "id": 597,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design an experimentation platform that supports thousands of concurrent A/B tests without interference.",
      "difficulty": 2
    },
    {
      "id": 598,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design a system for cross-sell/upsell recommendations at checkout.",
      "difficulty": 2
    },
    {
      "id": 599,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design a geo-fencing-based anomaly-detection system for a delivery/logistics platform.",
      "difficulty": 2
    },
    {
      "id": 600,
      "section": "Section 14 — Classic ML System Design",
      "question": "Design a system for personalized search query rewriting based on user history.",
      "difficulty": 2
    },
    {
      "id": 601,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "Explain the difference between online, batch, and streaming inference architectures.",
      "difficulty": 2
    },
    {
      "id": 602,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "What is model serving latency budget, and how do you allocate it across a multi-model pipeline?",
      "difficulty": 2
    },
    {
      "id": 603,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "Explain horizontal vs. vertical scaling for a model-serving cluster.",
      "difficulty": 2
    },
    {
      "id": 604,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "What is autoscaling based on, for GPU-backed inference services (queue depth, latency, utilization)?",
      "difficulty": 2
    },
    {
      "id": 605,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "Explain the tradeoff between serving many small models vs. one large multi-task model.",
      "difficulty": 2
    },
    {
      "id": 606,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "What is model warm-up, and why does cold-start latency matter for serverless inference?",
      "difficulty": 2
    },
    {
      "id": 607,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "Explain canary deployment and shadow deployment for model releases.",
      "difficulty": 2
    },
    {
      "id": 608,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "What is blue-green deployment, and how does it apply to model serving?",
      "difficulty": 2
    },
    {
      "id": 609,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "Explain how you'd design a rollback strategy for a bad model deployment.",
      "difficulty": 2
    },
    {
      "id": 610,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "What is batching at inference time and how does it trade off latency for throughput, and how does continuous batching (vLLM/SGLang-style) refine this specifically for LLM serving?",
      "difficulty": 2
    },
    {
      "id": 611,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "What is speculative decoding, and what hardware/latency profile benefits most from it?",
      "difficulty": 2
    },
    {
      "id": 612,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "Explain tensor parallelism vs. pipeline parallelism vs. data parallelism for serving very large models.",
      "difficulty": 2
    },
    {
      "id": 613,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "What is model sharding across GPUs, and when is it necessary vs. optional?",
      "difficulty": 2
    },
    {
      "id": 614,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "Explain the role of a model registry in a production ML platform.",
      "difficulty": 2
    },
    {
      "id": 615,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "What is a feature store's role at serving time (online store) vs. training time (offline store)?",
      "difficulty": 2
    },
    {
      "id": 616,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "Explain how you'd design feature freshness guarantees for real-time inference.",
      "difficulty": 2
    },
    {
      "id": 617,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "What is training/serving skew, and how do you detect and prevent it?",
      "difficulty": 2
    },
    {
      "id": 618,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "Explain multi-model serving frameworks (Triton, TorchServe, KServe, vLLM) and how you'd choose one.",
      "difficulty": 2
    },
    {
      "id": 619,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "What is GPU memory fragmentation, and how does paged attention address it for LLMs?",
      "difficulty": 2
    },
    {
      "id": 620,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "Explain the cost/latency/throughput tradeoff when choosing GPU type (A100 vs. H100 vs. L4, etc.) for serving.",
      "difficulty": 2
    },
    {
      "id": 621,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "What is model compilation (TensorRT, ONNX Runtime, torch.compile), and what speedups does it typically provide?",
      "difficulty": 2
    },
    {
      "id": 622,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "Explain the tradeoff between serving a quantized model vs. a full-precision one.",
      "difficulty": 2
    },
    {
      "id": 623,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "What is edge/on-device inference, and what constraints does it impose vs. cloud serving?",
      "difficulty": 2
    },
    {
      "id": 624,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "Explain how you'd design a hybrid edge-cloud inference architecture.",
      "difficulty": 2
    },
    {
      "id": 625,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "What is model versioning at serving time, and how do you support multiple concurrent versions safely?",
      "difficulty": 2
    },
    {
      "id": 626,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "Explain request coalescing/deduplication for identical concurrent inference requests.",
      "difficulty": 2
    },
    {
      "id": 627,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "What is a circuit breaker pattern, and how would you apply it to a flaky model-serving dependency?",
      "difficulty": 2
    },
    {
      "id": 628,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "Explain load shedding strategies when an inference service is overwhelmed.",
      "difficulty": 2
    },
    {
      "id": 629,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "What is the role of a feature/prompt cache in reducing serving cost?",
      "difficulty": 2
    },
    {
      "id": 630,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "Explain how you'd design multi-region serving for low global latency with data-residency constraints.",
      "difficulty": 2
    },
    {
      "id": 631,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "What is GPU utilization monitoring, and what metrics indicate you're over/under-provisioned?",
      "difficulty": 2
    },
    {
      "id": 632,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "Explain how you'd design cost-per-request observability across a fleet of models.",
      "difficulty": 2
    },
    {
      "id": 633,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "What is dynamic batching's failure mode (head-of-line blocking), and how do you mitigate it?",
      "difficulty": 2
    },
    {
      "id": 634,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "Explain the tradeoff between synchronous request-response and async job-queue architectures for long-running inference.",
      "difficulty": 2
    },
    {
      "id": 635,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "What is model warm pools, and how do they reduce cold-start latency in autoscaled environments?",
      "difficulty": 2
    },
    {
      "id": 636,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "Explain how you'd benchmark p50/p95/p99 latency for an inference service and why tail latency matters.",
      "difficulty": 2
    },
    {
      "id": 637,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "What is the role of a request timeout/deadline policy in a multi-hop inference pipeline?",
      "difficulty": 2
    },
    {
      "id": 638,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "Explain how you'd design graceful degradation (smaller model, cached answer) under peak load.",
      "difficulty": 2
    },
    {
      "id": 639,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "What is model ensembling's cost implication at serving time, and when is it still worth it?",
      "difficulty": 2
    },
    {
      "id": 640,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "Explain how you'd right-size GPU fleet capacity given unpredictable, spiky traffic.",
      "difficulty": 2
    },
    {
      "id": 641,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "What is the role of a service mesh in a microservices-based ML serving architecture?",
      "difficulty": 2
    },
    {
      "id": 642,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "Explain how you'd design zero-downtime model swaps in a high-traffic production system.",
      "difficulty": 2
    },
    {
      "id": 643,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "What is the tradeoff between self-hosting open-weight models vs. using a hosted API for serving?",
      "difficulty": 2
    },
    {
      "id": 644,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "Explain how you'd design a fallback chain across multiple model providers for reliability.",
      "difficulty": 2
    },
    {
      "id": 645,
      "section": "Section 15 — Model Serving & Inference Optimization",
      "question": "What is the impact of context length on both latency and cost at serving time, and how do you manage it?",
      "difficulty": 2
    },
    {
      "id": 646,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "How do you design a CI/CD pipeline for ML models, including automated eval gates before deploy?",
      "difficulty": 2
    },
    {
      "id": 647,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "What's your approach to versioning data, features, prompts, and models together for reproducibility?",
      "difficulty": 2
    },
    {
      "id": 648,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "How do you design rollback strategy for a bad model or prompt deployment?",
      "difficulty": 2
    },
    {
      "id": 649,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "Describe your approach to cost observability for GPU/inference spend across teams.",
      "difficulty": 2
    },
    {
      "id": 650,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "How do you scale training across multiple GPUs/nodes, and where does each parallelism strategy fail?",
      "difficulty": 2
    },
    {
      "id": 651,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "How do you handle model/prompt drift monitoring without ground-truth labels in production?",
      "difficulty": 2
    },
    {
      "id": 652,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "What does a good incident postmortem look like for an AI system failure?",
      "difficulty": 2
    },
    {
      "id": 653,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "Explain the difference between MLOps and LLMOps — what's genuinely new about LLMOps?",
      "difficulty": 2
    },
    {
      "id": 654,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "What is an experiment-tracking system (MLflow, Weights & Biases), and what should it capture?",
      "difficulty": 2
    },
    {
      "id": 655,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "Explain how you'd design a model registry with staged promotion (dev → staging → prod).",
      "difficulty": 2
    },
    {
      "id": 656,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "What is data versioning (DVC, LakeFS), and why does it matter for reproducibility?",
      "difficulty": 2
    },
    {
      "id": 657,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "Explain how you'd design automated retraining triggers based on drift detection.",
      "difficulty": 2
    },
    {
      "id": 658,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "What is champion/challenger testing in a production ML system?",
      "difficulty": 2
    },
    {
      "id": 659,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "Explain how you'd design a feature store's write path (streaming) vs. read path (low-latency serving).",
      "difficulty": 2
    },
    {
      "id": 660,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "What is data validation (Great Expectations, TFDV), and where does it fit in the pipeline?",
      "difficulty": 2
    },
    {
      "id": 661,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "Explain how you'd design schema evolution handling for a long-lived feature pipeline.",
      "difficulty": 2
    },
    {
      "id": 662,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "What is the role of a model card, and what should it document?",
      "difficulty": 2
    },
    {
      "id": 663,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "Explain how you'd design an automated eval suite that runs on every prompt or model change.",
      "difficulty": 2
    },
    {
      "id": 664,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "What is LLM-as-judge evaluation, and what are its known biases/limitations?",
      "difficulty": 2
    },
    {
      "id": 665,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "Explain how you'd combine offline evals, online A/B tests, and human review into one eval strategy.",
      "difficulty": 2
    },
    {
      "id": 666,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "What is a golden dataset, and how do you build and maintain one for regression testing?",
      "difficulty": 2
    },
    {
      "id": 667,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "Explain how you'd detect silent quality regressions in an LLM feature after a provider's model update.",
      "difficulty": 2
    },
    {
      "id": 668,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "What is prompt/model shadow testing, and how does it de-risk changes before full rollout?",
      "difficulty": 2
    },
    {
      "id": 669,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "Explain how you'd instrument token-level cost tracking across a multi-step agent pipeline.",
      "difficulty": 2
    },
    {
      "id": 670,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "What is the role of tracing (e.g., OpenTelemetry-style spans) in debugging a multi-hop LLM pipeline?",
      "difficulty": 2
    },
    {
      "id": 671,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "Explain how you'd design alerting thresholds for LLM quality metrics without excessive noise.",
      "difficulty": 2
    },
    {
      "id": 672,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "What is a feedback loop, and how would you design one that captures user corrections for future fine-tuning?",
      "difficulty": 2
    },
    {
      "id": 673,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "Explain how you'd handle a security incident where an LLM leaked sensitive data in its output.",
      "difficulty": 2
    },
    {
      "id": 674,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "What is your strategy for managing API key/credential rotation across many LLM-integrated services?",
      "difficulty": 2
    },
    {
      "id": 675,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "Explain how you'd design capacity planning for GPU clusters supporting both training and inference workloads.",
      "difficulty": 2
    },
    {
      "id": 676,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "What is spot/preemptible instance usage for training, and how do you handle interruption gracefully?",
      "difficulty": 2
    },
    {
      "id": 677,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "Explain checkpointing strategy for long-running distributed training jobs.",
      "difficulty": 2
    },
    {
      "id": 678,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "What is gradient accumulation, and when would you use it over increasing batch size directly?",
      "difficulty": 2
    },
    {
      "id": 679,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "Explain how you'd design a data pipeline for continuous fine-tuning from production feedback.",
      "difficulty": 2
    },
    {
      "id": 680,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "What is catastrophic forgetting risk when continuously fine-tuning a production model, and how do you guard against it?",
      "difficulty": 2
    },
    {
      "id": 681,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "Explain how you'd set up canary evaluation for a fine-tuned model before full rollout.",
      "difficulty": 2
    },
    {
      "id": 682,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "What is the role of synthetic data in LLMOps, and what are its risks (model collapse, bias amplification)?",
      "difficulty": 2
    },
    {
      "id": 683,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "Explain how you'd design cost attribution/chargeback for LLM usage across business units.",
      "difficulty": 2
    },
    {
      "id": 684,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "What is a \"kill switch\" for an AI feature, and how would you design one to be reliably fast?",
      "difficulty": 2
    },
    {
      "id": 685,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "Explain how you'd manage secrets and PII scrubbing in logs collected from LLM interactions.",
      "difficulty": 2
    },
    {
      "id": 686,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "What is the role of a feature-flagging system in safely rolling out AI features?",
      "difficulty": 2
    },
    {
      "id": 687,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "Explain how you'd design multi-environment parity (dev/staging/prod) for an LLM pipeline with external API dependencies.",
      "difficulty": 2
    },
    {
      "id": 688,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "What is dataset contamination, and how do you check whether your eval set leaked into training data?",
      "difficulty": 2
    },
    {
      "id": 689,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "Explain how you'd design a reproducible fine-tuning pipeline (seeded, versioned, containerized).",
      "difficulty": 2
    },
    {
      "id": 690,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "What is the role of infrastructure-as-code (Terraform) in managing ML platform environments?",
      "difficulty": 2
    },
    {
      "id": 691,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "Explain how you'd design blue/green rollout specifically for a fine-tuned LLM checkpoint.",
      "difficulty": 2
    },
    {
      "id": 692,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "What is model deprecation planning, and how do you sunset an old model version safely?",
      "difficulty": 2
    },
    {
      "id": 693,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "Explain how you'd design SLOs (service-level objectives) for an LLM-powered API.",
      "difficulty": 2
    },
    {
      "id": 694,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "What is error budgeting, and how would you apply it to an AI feature's reliability target?",
      "difficulty": 2
    },
    {
      "id": 695,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "Explain how you'd design an on-call runbook for an LLM-serving outage.",
      "difficulty": 2
    },
    {
      "id": 696,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "What is the role of synthetic monitoring (scheduled test queries) for catching silent degradation?",
      "difficulty": 2
    },
    {
      "id": 697,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "Explain how you'd handle a scenario where your evaluation metrics look good but users report poor quality.",
      "difficulty": 2
    },
    {
      "id": 698,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "What is the build vs. buy decision framework for MLOps tooling (managed platform vs. custom stack)?",
      "difficulty": 2
    },
    {
      "id": 699,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "Explain how you'd design data lineage tracking from raw source through to a deployed model's predictions.",
      "difficulty": 2
    },
    {
      "id": 700,
      "section": "Section 16 — LLMOps & MLOps",
      "question": "What is the role of a \"model risk\" review board in a regulated enterprise, and what would you present to it?",
      "difficulty": 2
    },
    {
      "id": 701,
      "section": "Section 17 — Feature Stores & Feature Engineering",
      "question": "What is a feature store, and why do training-serving consistency issues arise without one?",
      "difficulty": 2
    },
    {
      "id": 702,
      "section": "Section 17 — Feature Stores & Feature Engineering",
      "question": "Explain the difference between an online (low-latency) and offline (batch) feature store.",
      "difficulty": 2
    },
    {
      "id": 703,
      "section": "Section 17 — Feature Stores & Feature Engineering",
      "question": "What is point-in-time correctness, and why does it matter for avoiding label leakage?",
      "difficulty": 2
    },
    {
      "id": 704,
      "section": "Section 17 — Feature Stores & Feature Engineering",
      "question": "Explain feature versioning and how you'd roll out a new feature definition safely.",
      "difficulty": 2
    },
    {
      "id": 705,
      "section": "Section 17 — Feature Stores & Feature Engineering",
      "question": "What is feature freshness, and how do you monitor for stale features reaching a model?",
      "difficulty": 2
    },
    {
      "id": 706,
      "section": "Section 17 — Feature Stores & Feature Engineering",
      "question": "Explain how you'd design feature backfills for a newly added feature.",
      "difficulty": 2
    },
    {
      "id": 707,
      "section": "Section 17 — Feature Stores & Feature Engineering",
      "question": "What is entity resolution in the context of joining features across multiple data sources?",
      "difficulty": 2
    },
    {
      "id": 708,
      "section": "Section 17 — Feature Stores & Feature Engineering",
      "question": "Explain how streaming features (e.g., Kafka-based) differ architecturally from batch features.",
      "difficulty": 2
    },
    {
      "id": 709,
      "section": "Section 17 — Feature Stores & Feature Engineering",
      "question": "What is feature reuse across teams, and what governance is needed to prevent feature sprawl?",
      "difficulty": 2
    },
    {
      "id": 710,
      "section": "Section 17 — Feature Stores & Feature Engineering",
      "question": "Explain how you'd detect and handle a feature pipeline silently producing null/default values.",
      "difficulty": 2
    },
    {
      "id": 711,
      "section": "Section 17 — Feature Stores & Feature Engineering",
      "question": "What is target leakage in feature engineering, and give a concrete example.",
      "difficulty": 2
    },
    {
      "id": 712,
      "section": "Section 17 — Feature Stores & Feature Engineering",
      "question": "Explain binning/discretization and when it helps a model vs. when it discards useful signal.",
      "difficulty": 2
    },
    {
      "id": 713,
      "section": "Section 17 — Feature Stores & Feature Engineering",
      "question": "What is feature crossing, and when does it help linear models capture non-linear relationships?",
      "difficulty": 2
    },
    {
      "id": 714,
      "section": "Section 17 — Feature Stores & Feature Engineering",
      "question": "Explain how you'd engineer features from time-series data (lags, rolling windows, seasonality).",
      "difficulty": 2
    },
    {
      "id": 715,
      "section": "Section 17 — Feature Stores & Feature Engineering",
      "question": "What is embedding-based feature engineering for high-cardinality categorical variables?",
      "difficulty": 2
    },
    {
      "id": 716,
      "section": "Section 17 — Feature Stores & Feature Engineering",
      "question": "Explain how you'd handle missing data across different mechanisms (MCAR, MAR, MNAR).",
      "difficulty": 2
    },
    {
      "id": 717,
      "section": "Section 17 — Feature Stores & Feature Engineering",
      "question": "What is feature importance, and compare model-based (SHAP) vs. permutation-based methods.",
      "difficulty": 2
    },
    {
      "id": 718,
      "section": "Section 17 — Feature Stores & Feature Engineering",
      "question": "Explain how you'd design feature monitoring dashboards for a large production feature set.",
      "difficulty": 2
    },
    {
      "id": 719,
      "section": "Section 17 — Feature Stores & Feature Engineering",
      "question": "What is the cost tradeoff of computing expensive features in real time vs. precomputing them?",
      "difficulty": 2
    },
    {
      "id": 720,
      "section": "Section 17 — Feature Stores & Feature Engineering",
      "question": "Explain how you'd design a feature store to support both classic ML and LLM-based (retrieval) features.",
      "difficulty": 2
    },
    {
      "id": 721,
      "section": "Section 17 — Feature Stores & Feature Engineering",
      "question": "What is data skew between training and production feature distributions, and how do you catch it early?",
      "difficulty": 2
    },
    {
      "id": 722,
      "section": "Section 17 — Feature Stores & Feature Engineering",
      "question": "Explain how you'd design feature access control for sensitive attributes (e.g., protected classes).",
      "difficulty": 2
    },
    {
      "id": 723,
      "section": "Section 17 — Feature Stores & Feature Engineering",
      "question": "What is a feature pipeline's testing strategy — unit tests, integration tests, data contract tests?",
      "difficulty": 2
    },
    {
      "id": 724,
      "section": "Section 17 — Feature Stores & Feature Engineering",
      "question": "Explain how you'd migrate a legacy feature pipeline to a new feature store without breaking production models.",
      "difficulty": 2
    },
    {
      "id": 725,
      "section": "Section 17 — Feature Stores & Feature Engineering",
      "question": "What is the role of a feature catalog/discovery tool for a large ML organization?",
      "difficulty": 2
    },
    {
      "id": 726,
      "section": "Section 18 — Data Engineering for AI",
      "question": "How do you build data governance for AI (lineage, quality validation, access control) as a foundation, not an afterthought?",
      "difficulty": 2
    },
    {
      "id": 727,
      "section": "Section 18 — Data Engineering for AI",
      "question": "Explain the difference between a data warehouse, data lake, and lakehouse architecture.",
      "difficulty": 2
    },
    {
      "id": 728,
      "section": "Section 18 — Data Engineering for AI",
      "question": "What is Apache Spark's role in large-scale data processing for ML pipelines?",
      "difficulty": 2
    },
    {
      "id": 729,
      "section": "Section 18 — Data Engineering for AI",
      "question": "Explain Apache Kafka's role in streaming data pipelines feeding real-time features or RAG indexes.",
      "difficulty": 2
    },
    {
      "id": 730,
      "section": "Section 18 — Data Engineering for AI",
      "question": "What is Apache Airflow used for, and how would you design DAGs for a complex ML pipeline?",
      "difficulty": 2
    },
    {
      "id": 731,
      "section": "Section 18 — Data Engineering for AI",
      "question": "Explain dbt's role in the modern data stack, and how it differs from traditional ETL.",
      "difficulty": 2
    },
    {
      "id": 732,
      "section": "Section 18 — Data Engineering for AI",
      "question": "What is Apache Iceberg / Delta Lake, and why do table formats matter for large-scale analytics?",
      "difficulty": 2
    },
    {
      "id": 733,
      "section": "Section 18 — Data Engineering for AI",
      "question": "Explain the medallion architecture (bronze/silver/gold) for data lakehouses.",
      "difficulty": 2
    },
    {
      "id": 734,
      "section": "Section 18 — Data Engineering for AI",
      "question": "What is schema-on-read vs. schema-on-write, and when does each make sense?",
      "difficulty": 2
    },
    {
      "id": 735,
      "section": "Section 18 — Data Engineering for AI",
      "question": "Explain data partitioning strategies for large-scale query performance.",
      "difficulty": 2
    },
    {
      "id": 736,
      "section": "Section 18 — Data Engineering for AI",
      "question": "What is data deduplication at scale, and what algorithms/approaches are used?",
      "difficulty": 2
    },
    {
      "id": 737,
      "section": "Section 18 — Data Engineering for AI",
      "question": "Explain how you'd design a data pipeline for ingesting and cleaning unstructured documents at scale for RAG.",
      "difficulty": 2
    },
    {
      "id": 738,
      "section": "Section 18 — Data Engineering for AI",
      "question": "What is document parsing's biggest challenge (PDFs, scanned images, tables), and how do you handle it robustly?",
      "difficulty": 2
    },
    {
      "id": 739,
      "section": "Section 18 — Data Engineering for AI",
      "question": "Explain OCR pipeline design for scanned document ingestion.",
      "difficulty": 2
    },
    {
      "id": 740,
      "section": "Section 18 — Data Engineering for AI",
      "question": "What is data lineage, and how would you implement it across a multi-stage pipeline?",
      "difficulty": 2
    },
    {
      "id": 741,
      "section": "Section 18 — Data Engineering for AI",
      "question": "Explain change data capture (CDC) and its role in keeping downstream systems in sync.",
      "difficulty": 2
    },
    {
      "id": 742,
      "section": "Section 18 — Data Engineering for AI",
      "question": "What is idempotency in data pipelines, and why does it matter for reliability?",
      "difficulty": 2
    },
    {
      "id": 743,
      "section": "Section 18 — Data Engineering for AI",
      "question": "Explain exactly-once vs. at-least-once processing semantics in streaming pipelines.",
      "difficulty": 2
    },
    {
      "id": 744,
      "section": "Section 18 — Data Engineering for AI",
      "question": "What is data quality validation (Great Expectations style), and where should it run in the pipeline?",
      "difficulty": 2
    },
    {
      "id": 745,
      "section": "Section 18 — Data Engineering for AI",
      "question": "Explain how you'd design a data pipeline SLA (freshness, completeness, accuracy).",
      "difficulty": 2
    },
    {
      "id": 746,
      "section": "Section 18 — Data Engineering for AI",
      "question": "What is a data contract, and how does it prevent breaking changes between producer and consumer teams?",
      "difficulty": 2
    },
    {
      "id": 747,
      "section": "Section 18 — Data Engineering for AI",
      "question": "Explain how you'd design PII detection and redaction as an automated step in an ingestion pipeline.",
      "difficulty": 2
    },
    {
      "id": 748,
      "section": "Section 18 — Data Engineering for AI",
      "question": "What is data cataloging, and how does it help discoverability across a large data platform?",
      "difficulty": 2
    },
    {
      "id": 749,
      "section": "Section 18 — Data Engineering for AI",
      "question": "Explain how you'd handle schema drift from an upstream source system.",
      "difficulty": 2
    },
    {
      "id": 750,
      "section": "Section 18 — Data Engineering for AI",
      "question": "What is backpressure in a streaming pipeline, and how do you handle it gracefully?",
      "difficulty": 2
    },
    {
      "id": 751,
      "section": "Section 18 — Data Engineering for AI",
      "question": "Explain how you'd design a pipeline to deduplicate and merge multi-source customer data (identity resolution).",
      "difficulty": 2
    },
    {
      "id": 752,
      "section": "Section 18 — Data Engineering for AI",
      "question": "What is the tradeoff between row-based and columnar storage formats (Parquet, ORC)?",
      "difficulty": 2
    },
    {
      "id": 753,
      "section": "Section 18 — Data Engineering for AI",
      "question": "Explain how you'd design incremental processing to avoid reprocessing an entire dataset on each run.",
      "difficulty": 2
    },
    {
      "id": 754,
      "section": "Section 18 — Data Engineering for AI",
      "question": "What is data mesh, and how does it differ from a centralized data platform model?",
      "difficulty": 2
    },
    {
      "id": 755,
      "section": "Section 18 — Data Engineering for AI",
      "question": "Explain how you'd design multi-region data replication with consistency and residency requirements.",
      "difficulty": 2
    },
    {
      "id": 756,
      "section": "Section 18 — Data Engineering for AI",
      "question": "What is data retention policy design, and how does it intersect with regulatory requirements (GDPR right to erasure)?",
      "difficulty": 2
    },
    {
      "id": 757,
      "section": "Section 18 — Data Engineering for AI",
      "question": "Explain how you'd design a pipeline that ingests real-time events for both analytics and online feature serving.",
      "difficulty": 2
    },
    {
      "id": 758,
      "section": "Section 18 — Data Engineering for AI",
      "question": "What is a data quality \"circuit breaker\" that halts a pipeline before bad data reaches production models?",
      "difficulty": 2
    },
    {
      "id": 759,
      "section": "Section 18 — Data Engineering for AI",
      "question": "Explain your approach to cost optimization for large-scale data processing (spot instances, partition pruning, caching).",
      "difficulty": 2
    },
    {
      "id": 760,
      "section": "Section 18 — Data Engineering for AI",
      "question": "What is geospatial data processing (H3, PostGIS), and where does it intersect with AI systems?",
      "difficulty": 2
    },
    {
      "id": 761,
      "section": "Section 18 — Data Engineering for AI",
      "question": "Explain how you'd design a pipeline to continuously refresh a RAG corpus from live document sources.",
      "difficulty": 2
    },
    {
      "id": 762,
      "section": "Section 18 — Data Engineering for AI",
      "question": "What is the role of a metadata store in a modern data platform?",
      "difficulty": 2
    },
    {
      "id": 763,
      "section": "Section 18 — Data Engineering for AI",
      "question": "Explain how you'd design data access auditing for compliance purposes.",
      "difficulty": 2
    },
    {
      "id": 764,
      "section": "Section 18 — Data Engineering for AI",
      "question": "What is the tradeoff between ELT and ETL in a modern cloud data stack?",
      "difficulty": 2
    },
    {
      "id": 765,
      "section": "Section 18 — Data Engineering for AI",
      "question": "Explain how you'd design disaster recovery for a mission-critical data pipeline.",
      "difficulty": 2
    },
    {
      "id": 766,
      "section": "Section 19 — Cloud ML Platforms",
      "question": "Compare AWS SageMaker, Google Vertex AI, and Azure ML at a high level — when would you choose each?",
      "difficulty": 2
    },
    {
      "id": 767,
      "section": "Section 19 — Cloud ML Platforms",
      "question": "Explain SageMaker's training job vs. endpoint vs. batch transform — when to use each.",
      "difficulty": 2
    },
    {
      "id": 768,
      "section": "Section 19 — Cloud ML Platforms",
      "question": "What is Vertex AI Pipelines, and how does it compare to Airflow for ML workflow orchestration?",
      "difficulty": 2
    },
    {
      "id": 769,
      "section": "Section 19 — Cloud ML Platforms",
      "question": "Explain Azure ML's managed endpoints and how they support blue/green deployment.",
      "difficulty": 2
    },
    {
      "id": 770,
      "section": "Section 19 — Cloud ML Platforms",
      "question": "What is a managed feature store offering (SageMaker Feature Store, Vertex AI Feature Store), and its tradeoffs vs. self-hosted?",
      "difficulty": 2
    },
    {
      "id": 771,
      "section": "Section 19 — Cloud ML Platforms",
      "question": "Explain spot/preemptible instance strategies across AWS, GCP, and Azure for cost-efficient training.",
      "difficulty": 2
    },
    {
      "id": 772,
      "section": "Section 19 — Cloud ML Platforms",
      "question": "What is multi-cloud ML architecture, and what are the real costs (not just benefits) of pursuing it?",
      "difficulty": 2
    },
    {
      "id": 773,
      "section": "Section 19 — Cloud ML Platforms",
      "question": "Explain how you'd design IAM/access control for an ML platform spanning multiple cloud accounts.",
      "difficulty": 2
    },
    {
      "id": 774,
      "section": "Section 19 — Cloud ML Platforms",
      "question": "What is a managed vector search offering (e.g., Vertex AI Vector Search), and how does it compare to third-party vector DBs?",
      "difficulty": 2
    },
    {
      "id": 775,
      "section": "Section 19 — Cloud ML Platforms",
      "question": "Explain how you'd design cost governance/budgets across cloud ML services for a large organization.",
      "difficulty": 2
    },
    {
      "id": 776,
      "section": "Section 19 — Cloud ML Platforms",
      "question": "What is serverless inference (e.g., SageMaker Serverless Inference), and where does it fall short for LLM workloads?",
      "difficulty": 2
    },
    {
      "id": 777,
      "section": "Section 19 — Cloud ML Platforms",
      "question": "Explain how you'd design a hybrid on-prem/cloud ML architecture for a data-residency-constrained enterprise.",
      "difficulty": 2
    },
    {
      "id": 778,
      "section": "Section 19 — Cloud ML Platforms",
      "question": "What is a model garden/model hub offering, and how does it fit into your model-selection process?",
      "difficulty": 2
    },
    {
      "id": 779,
      "section": "Section 19 — Cloud ML Platforms",
      "question": "Explain how cloud-native autoscaling (e.g., Kubernetes HPA, cloud-specific autoscalers) applies to GPU inference workloads.",
      "difficulty": 2
    },
    {
      "id": 780,
      "section": "Section 19 — Cloud ML Platforms",
      "question": "What is the tradeoff between managed MLOps tooling (SageMaker Pipelines) and open-source (Kubeflow, MLflow) alternatives?",
      "difficulty": 2
    },
    {
      "id": 781,
      "section": "Section 19 — Cloud ML Platforms",
      "question": "Explain how you'd design cross-cloud disaster recovery for a critical ML service.",
      "difficulty": 2
    },
    {
      "id": 782,
      "section": "Section 19 — Cloud ML Platforms",
      "question": "What is egress cost, and how does it factor into a multi-cloud or hybrid architecture decision?",
      "difficulty": 2
    },
    {
      "id": 783,
      "section": "Section 19 — Cloud ML Platforms",
      "question": "Explain how you'd evaluate a cloud provider's GPU availability/quota constraints when planning a large training run.",
      "difficulty": 2
    },
    {
      "id": 784,
      "section": "Section 19 — Cloud ML Platforms",
      "question": "What is a private endpoint / VPC peering, and why does it matter for securing model-serving traffic?",
      "difficulty": 2
    },
    {
      "id": 785,
      "section": "Section 19 — Cloud ML Platforms",
      "question": "Explain how you'd design cost allocation tags/labels across cloud ML resources for chargeback reporting.",
      "difficulty": 2
    },
    {
      "id": 786,
      "section": "Section 19 — Cloud ML Platforms",
      "question": "What is the role of a cloud-native secrets manager in securing API keys for third-party LLM providers?",
      "difficulty": 2
    },
    {
      "id": 787,
      "section": "Section 19 — Cloud ML Platforms",
      "question": "Explain how you'd benchmark cloud GPU instance types for a specific inference workload before committing.",
      "difficulty": 2
    },
    {
      "id": 788,
      "section": "Section 19 — Cloud ML Platforms",
      "question": "What is reserved capacity/committed use discounting, and how would you plan for it given uncertain AI demand?",
      "difficulty": 2
    },
    {
      "id": 789,
      "section": "Section 19 — Cloud ML Platforms",
      "question": "Explain how you'd design a cloud cost anomaly detection system specifically for GPU spend.",
      "difficulty": 2
    },
    {
      "id": 790,
      "section": "Section 19 — Cloud ML Platforms",
      "question": "What is the tradeoff between using a cloud provider's native LLM API (Bedrock, Vertex AI, Azure OpenAI) vs. calling the model provider directly?",
      "difficulty": 2
    },
    {
      "id": 791,
      "section": "Section 19 — Cloud ML Platforms",
      "question": "Explain data residency and sovereignty requirements and how they shape cloud region selection for AI workloads.",
      "difficulty": 2
    },
    {
      "id": 792,
      "section": "Section 19 — Cloud ML Platforms",
      "question": "What is autoscaling cold-start latency across different cloud compute options (serverless vs. VM vs. Kubernetes)?",
      "difficulty": 2
    },
    {
      "id": 793,
      "section": "Section 19 — Cloud ML Platforms",
      "question": "Explain how you'd design a migration plan to move an ML platform from one cloud to another with minimal downtime.",
      "difficulty": 2
    },
    {
      "id": 794,
      "section": "Section 19 — Cloud ML Platforms",
      "question": "What is the role of managed Kubernetes (EKS/GKE/AKS) in hosting a self-managed model-serving layer?",
      "difficulty": 2
    },
    {
      "id": 795,
      "section": "Section 19 — Cloud ML Platforms",
      "question": "Explain how you'd choose between fully managed AI services and building your own on raw compute for cost/control tradeoffs.",
      "difficulty": 2
    },
    {
      "id": 796,
      "section": "Section 20 — DevOps & Infrastructure for AI",
      "question": "Explain Docker's role in packaging ML models for reproducible deployment.",
      "difficulty": 2
    },
    {
      "id": 797,
      "section": "Section 20 — DevOps & Infrastructure for AI",
      "question": "What is Kubernetes, and how does it orchestrate GPU-backed inference workloads?",
      "difficulty": 2
    },
    {
      "id": 798,
      "section": "Section 20 — DevOps & Infrastructure for AI",
      "question": "Explain Helm's role in managing complex Kubernetes deployments for an ML platform.",
      "difficulty": 2
    },
    {
      "id": 799,
      "section": "Section 20 — DevOps & Infrastructure for AI",
      "question": "What is Terraform, and how would you use it to manage ML infrastructure as code?",
      "difficulty": 2
    },
    {
      "id": 800,
      "section": "Section 20 — DevOps & Infrastructure for AI",
      "question": "Explain GitHub Actions (or similar CI/CD) for automating model testing and deployment.",
      "difficulty": 2
    },
    {
      "id": 801,
      "section": "Section 20 — DevOps & Infrastructure for AI",
      "question": "What is infrastructure drift, and how do you detect/prevent it in an ML platform's cloud resources?",
      "difficulty": 2
    },
    {
      "id": 802,
      "section": "Section 20 — DevOps & Infrastructure for AI",
      "question": "Explain how you'd design end-to-end testing for an AI system (unit, integration, and LLM-specific eval tests).",
      "difficulty": 2
    },
    {
      "id": 803,
      "section": "Section 20 — DevOps & Infrastructure for AI",
      "question": "What is GitOps, and how does it apply to managing ML deployment configurations?",
      "difficulty": 2
    },
    {
      "id": 804,
      "section": "Section 20 — DevOps & Infrastructure for AI",
      "question": "Explain how you'd containerize a GPU-dependent inference service correctly (CUDA versions, drivers).",
      "difficulty": 2
    },
    {
      "id": 805,
      "section": "Section 20 — DevOps & Infrastructure for AI",
      "question": "What is a service mesh (Istio/Linkerd), and when does it add value to an ML microservices architecture?",
      "difficulty": 2
    },
    {
      "id": 806,
      "section": "Section 20 — DevOps & Infrastructure for AI",
      "question": "Explain how you'd design secrets management for API keys used by dozens of AI-powered services.",
      "difficulty": 2
    },
    {
      "id": 807,
      "section": "Section 20 — DevOps & Infrastructure for AI",
      "question": "What is chaos engineering, and how would you apply it to test an AI system's resilience?",
      "difficulty": 2
    },
    {
      "id": 808,
      "section": "Section 20 — DevOps & Infrastructure for AI",
      "question": "Explain how you'd design health checks and readiness probes for a model-serving pod.",
      "difficulty": 2
    },
    {
      "id": 809,
      "section": "Section 20 — DevOps & Infrastructure for AI",
      "question": "What is horizontal pod autoscaling based on custom metrics (e.g., queue depth) for GPU workloads?",
      "difficulty": 2
    },
    {
      "id": 810,
      "section": "Section 20 — DevOps & Infrastructure for AI",
      "question": "Explain how you'd design a CI pipeline that runs LLM evals as a merge-blocking gate.",
      "difficulty": 2
    },
    {
      "id": 811,
      "section": "Section 20 — DevOps & Infrastructure for AI",
      "question": "What is infrastructure cost tagging, and how do you enforce it across teams deploying AI services?",
      "difficulty": 2
    },
    {
      "id": 812,
      "section": "Section 20 — DevOps & Infrastructure for AI",
      "question": "Explain how you'd design network policies to restrict which services can call external LLM APIs.",
      "difficulty": 2
    },
    {
      "id": 813,
      "section": "Section 20 — DevOps & Infrastructure for AI",
      "question": "What is a private container registry's role in securing custom model images?",
      "difficulty": 2
    },
    {
      "id": 814,
      "section": "Section 20 — DevOps & Infrastructure for AI",
      "question": "Explain how you'd design blue/green infrastructure for zero-downtime GPU cluster upgrades.",
      "difficulty": 2
    },
    {
      "id": 815,
      "section": "Section 20 — DevOps & Infrastructure for AI",
      "question": "What is observability's three pillars (logs, metrics, traces), and how do they apply differently to LLM systems vs. traditional services?",
      "difficulty": 2
    },
    {
      "id": 816,
      "section": "Section 20 — DevOps & Infrastructure for AI",
      "question": "Explain how you'd design load testing specifically for an LLM-serving endpoint (accounting for variable response length).",
      "difficulty": 2
    },
    {
      "id": 817,
      "section": "Section 20 — DevOps & Infrastructure for AI",
      "question": "What is a service-level indicator (SLI) you'd track for an LLM API beyond standard latency/error rate?",
      "difficulty": 2
    },
    {
      "id": 818,
      "section": "Section 20 — DevOps & Infrastructure for AI",
      "question": "Explain how you'd design multi-tenant resource isolation (noisy-neighbor prevention) on a shared GPU cluster.",
      "difficulty": 2
    },
    {
      "id": 819,
      "section": "Section 20 — DevOps & Infrastructure for AI",
      "question": "What is the role of a feature flag system (LaunchDarkly-style) in progressively rolling out an AI feature?",
      "difficulty": 2
    },
    {
      "id": 820,
      "section": "Section 20 — DevOps & Infrastructure for AI",
      "question": "Explain how you'd design an incident response playbook specific to an AI system producing harmful output live.",
      "difficulty": 2
    },
    {
      "id": 821,
      "section": "Section 20 — DevOps & Infrastructure for AI",
      "question": "What is infrastructure capacity planning for bursty AI workloads (e.g., seasonal demand spikes)?",
      "difficulty": 2
    },
    {
      "id": 822,
      "section": "Section 20 — DevOps & Infrastructure for AI",
      "question": "Explain how you'd design cost-aware autoscaling that avoids runaway GPU spend from a traffic spike or bug.",
      "difficulty": 2
    },
    {
      "id": 823,
      "section": "Section 20 — DevOps & Infrastructure for AI",
      "question": "What is the role of a bastion host / private networking in securing access to model training infrastructure?",
      "difficulty": 2
    },
    {
      "id": 824,
      "section": "Section 20 — DevOps & Infrastructure for AI",
      "question": "Explain how you'd design backup and restore procedures for model checkpoints and vector indexes.",
      "difficulty": 2
    },
    {
      "id": 825,
      "section": "Section 20 — DevOps & Infrastructure for AI",
      "question": "What is the tradeoff between running inference on Kubernetes vs. a specialized serving platform (e.g., Ray Serve, BentoML)?",
      "difficulty": 2
    },
    {
      "id": 826,
      "section": "Section 20 — DevOps & Infrastructure for AI",
      "question": "Explain how you'd design your CI/CD to test prompt changes with the same rigor as code changes.",
      "difficulty": 2
    },
    {
      "id": 827,
      "section": "Section 20 — DevOps & Infrastructure for AI",
      "question": "What is dependency pinning's importance for reproducible ML environments, and how do you manage it at scale?",
      "difficulty": 2
    },
    {
      "id": 828,
      "section": "Section 20 — DevOps & Infrastructure for AI",
      "question": "Explain how you'd design a rollback mechanism for infrastructure-as-code changes affecting production inference.",
      "difficulty": 2
    },
    {
      "id": 829,
      "section": "Section 20 — DevOps & Infrastructure for AI",
      "question": "What is the role of a change-management/approval process for high-risk production AI deployments?",
      "difficulty": 2
    },
    {
      "id": 830,
      "section": "Section 20 — DevOps & Infrastructure for AI",
      "question": "Explain how you'd design monitoring dashboards that a non-technical on-call responder could use during an incident.",
      "difficulty": 2
    },
    {
      "id": 831,
      "section": "Section 21 — LLM Evaluation",
      "question": "Design an LLM evaluation system: offline suites, LLM-as-judge, online A/B, regression gates.",
      "difficulty": 3
    },
    {
      "id": 832,
      "section": "Section 21 — LLM Evaluation",
      "question": "Explain the difference between reference-based and reference-free evaluation for generative output.",
      "difficulty": 3
    },
    {
      "id": 833,
      "section": "Section 21 — LLM Evaluation",
      "question": "What is LLM-as-judge, and what biases does it have (position bias, verbosity bias, self-preference)?",
      "difficulty": 3
    },
    {
      "id": 834,
      "section": "Section 21 — LLM Evaluation",
      "question": "Explain how you'd calibrate an LLM judge against human ratings before trusting it at scale.",
      "difficulty": 3
    },
    {
      "id": 835,
      "section": "Section 21 — LLM Evaluation",
      "question": "What is a golden/regression test set, and how do you keep it representative as usage evolves?",
      "difficulty": 3
    },
    {
      "id": 836,
      "section": "Section 21 — LLM Evaluation",
      "question": "Explain pairwise comparison (A/B) evaluation vs. absolute scoring for generative outputs.",
      "difficulty": 3
    },
    {
      "id": 837,
      "section": "Section 21 — LLM Evaluation",
      "question": "What is task-specific evaluation (e.g., exact match for QA, code execution pass rate) vs. general-purpose eval?",
      "difficulty": 3
    },
    {
      "id": 838,
      "section": "Section 21 — LLM Evaluation",
      "question": "Explain how you'd design an evaluation harness for a multi-turn conversational agent.",
      "difficulty": 3
    },
    {
      "id": 839,
      "section": "Section 21 — LLM Evaluation",
      "question": "What is groundedness/faithfulness evaluation, and how do you measure it automatically?",
      "difficulty": 3
    },
    {
      "id": 840,
      "section": "Section 21 — LLM Evaluation",
      "question": "Explain how you'd detect hallucination systematically across a large volume of outputs.",
      "difficulty": 3
    },
    {
      "id": 841,
      "section": "Section 21 — LLM Evaluation",
      "question": "What is the role of human evaluation, and how do you design rubrics that reduce rater disagreement?",
      "difficulty": 3
    },
    {
      "id": 842,
      "section": "Section 21 — LLM Evaluation",
      "question": "Explain inter-rater reliability (e.g., Cohen's kappa) and why it matters for human eval pipelines.",
      "difficulty": 3
    },
    {
      "id": 843,
      "section": "Section 21 — LLM Evaluation",
      "question": "What is red-teaming, and how would you structure a red-team exercise for a new LLM feature?",
      "difficulty": 3
    },
    {
      "id": 844,
      "section": "Section 21 — LLM Evaluation",
      "question": "Explain how you'd build an adversarial test suite targeting known failure modes (bias, jailbreaks, factual errors).",
      "difficulty": 3
    },
    {
      "id": 845,
      "section": "Section 21 — LLM Evaluation",
      "question": "What is benchmark contamination, and how do you guard your eval set against it?",
      "difficulty": 3
    },
    {
      "id": 846,
      "section": "Section 21 — LLM Evaluation",
      "question": "Explain how you'd evaluate an agent's tool-use correctness separately from its final answer quality.",
      "difficulty": 3
    },
    {
      "id": 847,
      "section": "Section 21 — LLM Evaluation",
      "question": "What is cost-normalized evaluation (quality per dollar), and why does it matter for model selection?",
      "difficulty": 3
    },
    {
      "id": 848,
      "section": "Section 21 — LLM Evaluation",
      "question": "Explain how you'd design online evaluation (implicit signals: thumbs up/down, retry rate, session abandonment).",
      "difficulty": 3
    },
    {
      "id": 849,
      "section": "Section 21 — LLM Evaluation",
      "question": "What is the tradeoff between automated metrics (BLEU/ROUGE) and LLM-judge metrics for summarization quality?",
      "difficulty": 3
    },
    {
      "id": 850,
      "section": "Section 21 — LLM Evaluation",
      "question": "Explain how you'd evaluate factual consistency for a RAG system specifically.",
      "difficulty": 3
    },
    {
      "id": 851,
      "section": "Section 21 — LLM Evaluation",
      "question": "What is a \"canary eval,\" and how would you run it before a full model/prompt rollout?",
      "difficulty": 3
    },
    {
      "id": 852,
      "section": "Section 21 — LLM Evaluation",
      "question": "Explain how you'd design evaluation for safety-critical outputs (medical, legal, financial advice).",
      "difficulty": 3
    },
    {
      "id": 853,
      "section": "Section 21 — LLM Evaluation",
      "question": "What is the role of synthetic adversarial data generation in expanding an eval suite's coverage?",
      "difficulty": 3
    },
    {
      "id": 854,
      "section": "Section 21 — LLM Evaluation",
      "question": "Explain how you'd track evaluation metrics over time to catch slow, silent degradation.",
      "difficulty": 3
    },
    {
      "id": 855,
      "section": "Section 21 — LLM Evaluation",
      "question": "What is the difference between evaluating a model in isolation vs. evaluating the full product experience?",
      "difficulty": 3
    },
    {
      "id": 856,
      "section": "Section 21 — LLM Evaluation",
      "question": "Explain how you'd design evaluation for latency-sensitive tradeoffs (is a faster, slightly worse answer acceptable?).",
      "difficulty": 3
    },
    {
      "id": 857,
      "section": "Section 21 — LLM Evaluation",
      "question": "What is a rubric-based eval, and how do you translate subjective quality into a scorable rubric?",
      "difficulty": 3
    },
    {
      "id": 858,
      "section": "Section 21 — LLM Evaluation",
      "question": "Explain how you'd evaluate multilingual model performance fairly across languages with different resource levels.",
      "difficulty": 3
    },
    {
      "id": 859,
      "section": "Section 21 — LLM Evaluation",
      "question": "What is the role of eval-driven development, where evals are written before the feature is built?",
      "difficulty": 3
    },
    {
      "id": 860,
      "section": "Section 21 — LLM Evaluation",
      "question": "Explain how you'd evaluate an agent's efficiency (steps taken, cost) in addition to its correctness.",
      "difficulty": 3
    },
    {
      "id": 861,
      "section": "Section 21 — LLM Evaluation",
      "question": "What is the risk of over-optimizing for an eval metric (Goodhart's Law) in an LLM product?",
      "difficulty": 3
    },
    {
      "id": 862,
      "section": "Section 21 — LLM Evaluation",
      "question": "Explain how you'd structure eval ownership across teams (platform team vs. product team responsibilities).",
      "difficulty": 3
    },
    {
      "id": 863,
      "section": "Section 21 — LLM Evaluation",
      "question": "What is a shadow eval pipeline, and how does it run in parallel with production without affecting users?",
      "difficulty": 3
    },
    {
      "id": 864,
      "section": "Section 21 — LLM Evaluation",
      "question": "Explain how you'd design evaluation specifically for a code-generation feature (execution-based testing).",
      "difficulty": 3
    },
    {
      "id": 865,
      "section": "Section 21 — LLM Evaluation",
      "question": "What is your approach to evaluating an LLM feature when you have very little labeled data to start?",
      "difficulty": 3
    },
    {
      "id": 866,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "How do you design guardrails/safety filtering for both inputs and outputs, including jailbreak defense?",
      "difficulty": 3
    },
    {
      "id": 867,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "Explain prompt injection and the difference between direct and indirect (document-borne) injection.",
      "difficulty": 3
    },
    {
      "id": 868,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "What is a jailbreak, and how do techniques like role-play or encoding attacks attempt to bypass safety training?",
      "difficulty": 3
    },
    {
      "id": 869,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "Explain how you'd design defense-in-depth against prompt injection across multiple layers (input filtering, system prompt hardening, output validation).",
      "difficulty": 3
    },
    {
      "id": 870,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "What is PII detection/redaction in an LLM pipeline, and where should it run (pre-prompt, post-output, both)?",
      "difficulty": 3
    },
    {
      "id": 871,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "Explain how you'd design content moderation for both user inputs and model outputs.",
      "difficulty": 3
    },
    {
      "id": 872,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "What is a \"system prompt leak,\" and how do you defend against a user extracting it?",
      "difficulty": 3
    },
    {
      "id": 873,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "Explain how you'd design rate limiting to prevent abuse (scraping, automated attacks) of an LLM API.",
      "difficulty": 3
    },
    {
      "id": 874,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "What is adversarial robustness testing, and how would you structure it for a production LLM feature?",
      "difficulty": 3
    },
    {
      "id": 875,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "Explain how you'd handle a scenario where a retrieved document in a RAG pipeline contains malicious instructions.",
      "difficulty": 3
    },
    {
      "id": 876,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "What is the risk of an agent with tool access being manipulated into taking a harmful real-world action?",
      "difficulty": 3
    },
    {
      "id": 877,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "Explain how you'd design permission scoping for an agent's tools (least privilege).",
      "difficulty": 3
    },
    {
      "id": 878,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "What is data exfiltration risk via LLM output, and how do you mitigate it (e.g., markdown image rendering exploits)?",
      "difficulty": 3
    },
    {
      "id": 879,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "Explain how you'd design output filtering for toxic, biased, or otherwise harmful generated content.",
      "difficulty": 3
    },
    {
      "id": 880,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "What is model theft/extraction risk, and how do you mitigate it for a proprietary fine-tuned model?",
      "difficulty": 3
    },
    {
      "id": 881,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "Explain how you'd design logging that captures enough for security investigation without over-retaining sensitive data.",
      "difficulty": 3
    },
    {
      "id": 882,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "What is the risk of training-data poisoning, and how would you detect it in a fine-tuning pipeline?",
      "difficulty": 3
    },
    {
      "id": 883,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "Explain how you'd design an incident-response plan specifically for an LLM producing harmful content live in production.",
      "difficulty": 3
    },
    {
      "id": 884,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "What is differential privacy, and where might it apply to protecting training data used in fine-tuning?",
      "difficulty": 3
    },
    {
      "id": 885,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "Explain how you'd design access control so an internal RAG assistant never surfaces data a given user shouldn't see.",
      "difficulty": 3
    },
    {
      "id": 886,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "What is the OWASP Top 10 for LLM Applications, and which risks are most relevant to your architecture?",
      "difficulty": 3
    },
    {
      "id": 887,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "Explain how you'd test for excessive agency — an agent taking actions beyond its intended scope.",
      "difficulty": 3
    },
    {
      "id": 888,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "What is model supply-chain security, and how do you vet a third-party fine-tuned or open-weight model before deployment?",
      "difficulty": 3
    },
    {
      "id": 889,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "Explain how you'd design a bug-bounty or responsible-disclosure program specific to AI safety issues.",
      "difficulty": 3
    },
    {
      "id": 890,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "What is the risk of insecure output handling (e.g., LLM output executed as code without sanitization)?",
      "difficulty": 3
    },
    {
      "id": 891,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "Explain how you'd design guardrails that reject unsafe requests without being so strict they block legitimate use.",
      "difficulty": 3
    },
    {
      "id": 892,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "What is the tradeoff between client-side and server-side safety filtering?",
      "difficulty": 3
    },
    {
      "id": 893,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "Explain how you'd design a system to detect coordinated abuse (many accounts probing for jailbreaks).",
      "difficulty": 3
    },
    {
      "id": 894,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "What is watermarking for AI-generated content, and what are its current limitations?",
      "difficulty": 3
    },
    {
      "id": 895,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "Explain how you'd design safety evaluation specifically for a model deployed in a children's or education product.",
      "difficulty": 3
    },
    {
      "id": 896,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "What is the risk profile difference between a closed-model API and a self-hosted open-weight model from a security standpoint?",
      "difficulty": 3
    },
    {
      "id": 897,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "Explain how you'd design monitoring to detect a sudden spike in jailbreak attempts.",
      "difficulty": 3
    },
    {
      "id": 898,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "What is the role of a \"constitution\" or explicit policy document in shaping model behavior via system prompts or fine-tuning?",
      "difficulty": 3
    },
    {
      "id": 899,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "Explain how you'd handle conflicting requirements between user personalization and privacy protection.",
      "difficulty": 3
    },
    {
      "id": 900,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "What is secure multi-party computation, and is it relevant to any of your AI architecture decisions?",
      "difficulty": 3
    },
    {
      "id": 901,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "Explain how you'd design an audit trail for every action an autonomous agent takes in a production system.",
      "difficulty": 3
    },
    {
      "id": 902,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "What is the risk of model output being used to reconstruct sensitive training data (membership inference)?",
      "difficulty": 3
    },
    {
      "id": 903,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "Explain how you'd design a safety review process that gates new AI features before launch.",
      "difficulty": 3
    },
    {
      "id": 904,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "What is your approach to balancing user trust/transparency (e.g., disclosing AI use) against product friction?",
      "difficulty": 3
    },
    {
      "id": 905,
      "section": "Section 22 — Safety, Guardrails & LLM Security",
      "question": "Explain how you'd design a system for users to report unsafe or incorrect AI outputs, and how that feeds back into fixes.",
      "difficulty": 3
    },
    {
      "id": 906,
      "section": "Section 23 — Governance, Ethics & Responsible AI",
      "question": "How do you approach bias detection and mitigation in a model affecting real people (hiring, lending, moderation)?",
      "difficulty": 3
    },
    {
      "id": 907,
      "section": "Section 23 — Governance, Ethics & Responsible AI",
      "question": "Explain demographic parity, equalized odds, and equal opportunity as fairness definitions — why can't you satisfy all at once?",
      "difficulty": 3
    },
    {
      "id": 908,
      "section": "Section 23 — Governance, Ethics & Responsible AI",
      "question": "What is disparate impact, and how would you test a model for it before launch?",
      "difficulty": 3
    },
    {
      "id": 909,
      "section": "Section 23 — Governance, Ethics & Responsible AI",
      "question": "Explain how you'd design a fairness audit process for a high-stakes model.",
      "difficulty": 3
    },
    {
      "id": 910,
      "section": "Section 23 — Governance, Ethics & Responsible AI",
      "question": "What's your framework for deciding whether a use case needs human-in-the-loop review before action?",
      "difficulty": 3
    },
    {
      "id": 911,
      "section": "Section 23 — Governance, Ethics & Responsible AI",
      "question": "How do you evaluate a third-party model/vendor for compliance (SOC2, data residency, training-data usage)?",
      "difficulty": 3
    },
    {
      "id": 912,
      "section": "Section 23 — Governance, Ethics & Responsible AI",
      "question": "Explain how you'd handle PII/sensitive data through an LLM pipeline end to end (ingestion, prompts, logs, outputs).",
      "difficulty": 3
    },
    {
      "id": 913,
      "section": "Section 23 — Governance, Ethics & Responsible AI",
      "question": "What is explainability, and compare SHAP vs. LIME as post-hoc explanation methods.",
      "difficulty": 3
    },
    {
      "id": 914,
      "section": "Section 23 — Governance, Ethics & Responsible AI",
      "question": "Explain the difference between interpretability and explainability, and when each is required by regulation.",
      "difficulty": 3
    },
    {
      "id": 915,
      "section": "Section 23 — Governance, Ethics & Responsible AI",
      "question": "What is the EU AI Act's risk-based classification, and how would it affect your architecture decisions (at a high level)?",
      "difficulty": 3
    },
    {
      "id": 916,
      "section": "Section 23 — Governance, Ethics & Responsible AI",
      "question": "Explain how you'd design a model documentation process (model cards, datasheets for datasets) for audit readiness.",
      "difficulty": 3
    },
    {
      "id": 917,
      "section": "Section 23 — Governance, Ethics & Responsible AI",
      "question": "What is algorithmic accountability, and who should own it inside an organization (legal, product, engineering)?",
      "difficulty": 3
    },
    {
      "id": 918,
      "section": "Section 23 — Governance, Ethics & Responsible AI",
      "question": "Explain how you'd handle a discovered bias issue in a model already in production.",
      "difficulty": 3
    },
    {
      "id": 919,
      "section": "Section 23 — Governance, Ethics & Responsible AI",
      "question": "What is consent and data-usage transparency, and how does it apply to using customer data for model training?",
      "difficulty": 3
    },
    {
      "id": 920,
      "section": "Section 23 — Governance, Ethics & Responsible AI",
      "question": "Explain how you'd design a responsible-AI review board's intake process for new AI features.",
      "difficulty": 3
    },
    {
      "id": 921,
      "section": "Section 23 — Governance, Ethics & Responsible AI",
      "question": "What is the \"right to explanation,\" and how would you operationalize it for an automated decision system?",
      "difficulty": 3
    },
    {
      "id": 922,
      "section": "Section 23 — Governance, Ethics & Responsible AI",
      "question": "Explain how you'd balance model performance against fairness constraints when they conflict.",
      "difficulty": 3
    },
    {
      "id": 923,
      "section": "Section 23 — Governance, Ethics & Responsible AI",
      "question": "What is data minimization, and how does it apply to designing an LLM feature's data pipeline?",
      "difficulty": 3
    },
    {
      "id": 924,
      "section": "Section 23 — Governance, Ethics & Responsible AI",
      "question": "Explain how you'd design environmental-impact reporting (compute/energy) for large training runs.",
      "difficulty": 3
    },
    {
      "id": 925,
      "section": "Section 23 — Governance, Ethics & Responsible AI",
      "question": "What is the risk of automation bias — humans over-trusting AI recommendations — and how do you design against it?",
      "difficulty": 3
    },
    {
      "id": 926,
      "section": "Section 23 — Governance, Ethics & Responsible AI",
      "question": "Explain how you'd design a process for retiring/sunsetting a biased or harmful model responsibly.",
      "difficulty": 3
    },
    {
      "id": 927,
      "section": "Section 23 — Governance, Ethics & Responsible AI",
      "question": "What is synthetic data's role in privacy-preserving model development, and its limitations?",
      "difficulty": 3
    },
    {
      "id": 928,
      "section": "Section 23 — Governance, Ethics & Responsible AI",
      "question": "Explain how you'd handle a regulator's request to audit your AI system's decision-making.",
      "difficulty": 3
    },
    {
      "id": 929,
      "section": "Section 23 — Governance, Ethics & Responsible AI",
      "question": "What is the difference between \"fair\" and \"unbiased\" in a practical model-evaluation context?",
      "difficulty": 3
    },
    {
      "id": 930,
      "section": "Section 23 — Governance, Ethics & Responsible AI",
      "question": "Explain how you'd design informed-consent flows for users interacting with an AI system making consequential decisions.",
      "difficulty": 3
    },
    {
      "id": 931,
      "section": "Section 23 — Governance, Ethics & Responsible AI",
      "question": "What is model risk management (as used in financial services, e.g., SR 11-7), and how does it apply beyond banking?",
      "difficulty": 3
    },
    {
      "id": 932,
      "section": "Section 23 — Governance, Ethics & Responsible AI",
      "question": "Explain how you'd structure ongoing bias monitoring (not just pre-launch testing) for a production model.",
      "difficulty": 3
    },
    {
      "id": 933,
      "section": "Section 23 — Governance, Ethics & Responsible AI",
      "question": "What is the tension between personalization and privacy, and how would you resolve it architecturally?",
      "difficulty": 3
    },
    {
      "id": 934,
      "section": "Section 23 — Governance, Ethics & Responsible AI",
      "question": "Explain how you'd design a data-deletion/right-to-erasure pipeline that also removes influence from a trained model.",
      "difficulty": 3
    },
    {
      "id": 935,
      "section": "Section 23 — Governance, Ethics & Responsible AI",
      "question": "What is copyright/IP risk in generative AI output, and how would you mitigate it in a customer-facing product?",
      "difficulty": 3
    },
    {
      "id": 936,
      "section": "Section 23 — Governance, Ethics & Responsible AI",
      "question": "Explain how you'd handle attribution and licensing when using open-weight models with restrictive licenses.",
      "difficulty": 3
    },
    {
      "id": 937,
      "section": "Section 23 — Governance, Ethics & Responsible AI",
      "question": "What is the role of third-party AI audits, and when would you commission one?",
      "difficulty": 3
    },
    {
      "id": 938,
      "section": "Section 23 — Governance, Ethics & Responsible AI",
      "question": "Explain how you'd design escalation paths when an AI system's output could cause real-world harm.",
      "difficulty": 3
    },
    {
      "id": 939,
      "section": "Section 23 — Governance, Ethics & Responsible AI",
      "question": "What is stakeholder mapping for responsible AI governance (who needs a seat at the table)?",
      "difficulty": 3
    },
    {
      "id": 940,
      "section": "Section 23 — Governance, Ethics & Responsible AI",
      "question": "Explain how you'd build a culture where engineers proactively flag ethical concerns rather than staying silent.",
      "difficulty": 3
    },
    {
      "id": 941,
      "section": "Section 24 — Time Series & Forecasting",
      "question": "Explain the components of a time series: trend, seasonality, cyclicality, and noise.",
      "difficulty": 2
    },
    {
      "id": 942,
      "section": "Section 24 — Time Series & Forecasting",
      "question": "What is stationarity, and how do you test for it (ADF test)?",
      "difficulty": 2
    },
    {
      "id": 943,
      "section": "Section 24 — Time Series & Forecasting",
      "question": "Explain ARIMA and its components (AR, I, MA).",
      "difficulty": 2
    },
    {
      "id": 944,
      "section": "Section 24 — Time Series & Forecasting",
      "question": "What is exponential smoothing, and how does it differ from ARIMA?",
      "difficulty": 2
    },
    {
      "id": 945,
      "section": "Section 24 — Time Series & Forecasting",
      "question": "Explain Prophet's approach to forecasting and when it's preferable to classical methods.",
      "difficulty": 2
    },
    {
      "id": 946,
      "section": "Section 24 — Time Series & Forecasting",
      "question": "What is a rolling/expanding window validation strategy for time-series models, and why can't you use standard k-fold CV?",
      "difficulty": 2
    },
    {
      "id": 947,
      "section": "Section 24 — Time Series & Forecasting",
      "question": "Explain multivariate time-series forecasting and how it differs from univariate.",
      "difficulty": 2
    },
    {
      "id": 948,
      "section": "Section 24 — Time Series & Forecasting",
      "question": "What is a lag feature, and how do you choose which lags to include?",
      "difficulty": 2
    },
    {
      "id": 949,
      "section": "Section 24 — Time Series & Forecasting",
      "question": "Explain how transformer-based models (e.g., Temporal Fusion Transformer) apply to forecasting.",
      "difficulty": 2
    },
    {
      "id": 950,
      "section": "Section 24 — Time Series & Forecasting",
      "question": "What is concept drift specific to time series, and how do you detect a regime change?",
      "difficulty": 2
    },
    {
      "id": 951,
      "section": "Section 24 — Time Series & Forecasting",
      "question": "Explain how you'd handle missing timestamps or irregular sampling in a time-series dataset.",
      "difficulty": 2
    },
    {
      "id": 952,
      "section": "Section 24 — Time Series & Forecasting",
      "question": "What is backtesting, and how do you design it to avoid lookahead bias?",
      "difficulty": 2
    },
    {
      "id": 953,
      "section": "Section 24 — Time Series & Forecasting",
      "question": "Explain hierarchical forecasting (e.g., forecasting at SKU level that must reconcile to category level).",
      "difficulty": 2
    },
    {
      "id": 954,
      "section": "Section 24 — Time Series & Forecasting",
      "question": "What is anomaly detection in time series, and compare statistical vs. ML-based approaches.",
      "difficulty": 2
    },
    {
      "id": 955,
      "section": "Section 24 — Time Series & Forecasting",
      "question": "Explain how you'd choose a forecast horizon and its effect on model choice and uncertainty.",
      "difficulty": 2
    },
    {
      "id": 956,
      "section": "Section 24 — Time Series & Forecasting",
      "question": "What is prediction interval vs. point forecast, and why do stakeholders often need both?",
      "difficulty": 2
    },
    {
      "id": 957,
      "section": "Section 24 — Time Series & Forecasting",
      "question": "Explain how weather, holidays, or promotions would be incorporated as exogenous variables in a forecast model.",
      "difficulty": 2
    },
    {
      "id": 958,
      "section": "Section 24 — Time Series & Forecasting",
      "question": "What is the cold-start problem for forecasting a new product/SKU with no history?",
      "difficulty": 2
    },
    {
      "id": 959,
      "section": "Section 24 — Time Series & Forecasting",
      "question": "Explain how you'd evaluate forecast accuracy (MAPE, RMSE, WAPE) and their respective pitfalls.",
      "difficulty": 2
    },
    {
      "id": 960,
      "section": "Section 24 — Time Series & Forecasting",
      "question": "What is ensemble forecasting, and how do you combine multiple models' predictions robustly?",
      "difficulty": 2
    },
    {
      "id": 961,
      "section": "Section 25 — Recommender Systems",
      "question": "Explain collaborative filtering (user-based vs. item-based) and its cold-start weaknesses.",
      "difficulty": 2
    },
    {
      "id": 962,
      "section": "Section 25 — Recommender Systems",
      "question": "What is matrix factorization (e.g., ALS, SVD), and how does it scale to millions of users/items?",
      "difficulty": 2
    },
    {
      "id": 963,
      "section": "Section 25 — Recommender Systems",
      "question": "Explain content-based filtering and how it complements collaborative filtering in a hybrid system.",
      "difficulty": 2
    },
    {
      "id": 964,
      "section": "Section 25 — Recommender Systems",
      "question": "What is a two-tower model architecture for large-scale recommendation retrieval?",
      "difficulty": 2
    },
    {
      "id": 965,
      "section": "Section 25 — Recommender Systems",
      "question": "Explain the candidate-generation and ranking two-stage recommender architecture.",
      "difficulty": 2
    },
    {
      "id": 966,
      "section": "Section 25 — Recommender Systems",
      "question": "What is implicit feedback, and how do you train a model when you only have clicks, not explicit ratings?",
      "difficulty": 2
    },
    {
      "id": 967,
      "section": "Section 25 — Recommender Systems",
      "question": "Explain diversity and serendipity in recommendations, and how you'd measure/optimize for them alongside relevance.",
      "difficulty": 2
    },
    {
      "id": 968,
      "section": "Section 25 — Recommender Systems",
      "question": "What is exposure bias in recommender systems, and how does it create feedback loops that narrow content diversity?",
      "difficulty": 2
    },
    {
      "id": 969,
      "section": "Section 25 — Recommender Systems",
      "question": "Explain how you'd design a recommender system evaluation offline (NDCG, precision@k) vs. online (A/B, engagement).",
      "difficulty": 2
    },
    {
      "id": 970,
      "section": "Section 25 — Recommender Systems",
      "question": "What is session-based recommendation, and how does it differ from long-term user-profile-based recommendation?",
      "difficulty": 2
    },
    {
      "id": 971,
      "section": "Section 25 — Recommender Systems",
      "question": "Explain how graph neural networks are applied to recommendation (e.g., modeling user-item interaction graphs).",
      "difficulty": 2
    },
    {
      "id": 972,
      "section": "Section 25 — Recommender Systems",
      "question": "What is multi-objective recommendation (balancing engagement, revenue, diversity, fairness), and how do you weight objectives?",
      "difficulty": 2
    },
    {
      "id": 973,
      "section": "Section 25 — Recommender Systems",
      "question": "Explain how you'd handle the cold-start problem for a brand-new user with no interaction history.",
      "difficulty": 2
    },
    {
      "id": 974,
      "section": "Section 25 — Recommender Systems",
      "question": "What is real-time personalization, and what latency/infrastructure does it require versus batch-computed recommendations?",
      "difficulty": 2
    },
    {
      "id": 975,
      "section": "Section 25 — Recommender Systems",
      "question": "Explain how LLM-based re-ranking can be layered on top of a traditional recommendation pipeline.",
      "difficulty": 2
    },
    {
      "id": 976,
      "section": "Section 25 — Recommender Systems",
      "question": "What is popularity bias, and how do you correct for it without tanking overall engagement metrics?",
      "difficulty": 2
    },
    {
      "id": 977,
      "section": "Section 25 — Recommender Systems",
      "question": "Explain how you'd design an explanation (\"recommended because...\") feature for a recommender system.",
      "difficulty": 2
    },
    {
      "id": 978,
      "section": "Section 25 — Recommender Systems",
      "question": "What is negative sampling, and why is it necessary when training on implicit feedback at scale?",
      "difficulty": 2
    },
    {
      "id": 979,
      "section": "Section 25 — Recommender Systems",
      "question": "Explain how you'd design a recommender system to respect user-stated preferences/exclusions.",
      "difficulty": 2
    },
    {
      "id": 980,
      "section": "Section 25 — Recommender Systems",
      "question": "What is the feedback loop risk in recommender systems, and how would you audit for filter bubbles?",
      "difficulty": 2
    },
    {
      "id": 981,
      "section": "Section 26 — Coding & Algorithms for ML",
      "question": "Implement k-means clustering from scratch — what are the key steps and failure modes (empty clusters, bad init)?",
      "difficulty": 3
    },
    {
      "id": 982,
      "section": "Section 26 — Coding & Algorithms for ML",
      "question": "Implement logistic regression's gradient descent update from scratch.",
      "difficulty": 2
    },
    {
      "id": 983,
      "section": "Section 26 — Coding & Algorithms for ML",
      "question": "Write code to compute a confusion matrix and derive precision/recall/F1 from it.",
      "difficulty": 3
    },
    {
      "id": 984,
      "section": "Section 26 — Coding & Algorithms for ML",
      "question": "Implement a basic decision tree split (Gini or entropy) from scratch.",
      "difficulty": 3
    },
    {
      "id": 985,
      "section": "Section 26 — Coding & Algorithms for ML",
      "question": "Write a function to compute cosine similarity between two vectors efficiently at scale.",
      "difficulty": 1
    },
    {
      "id": 986,
      "section": "Section 26 — Coding & Algorithms for ML",
      "question": "Implement a simple k-nearest-neighbors classifier from scratch.",
      "difficulty": 3
    },
    {
      "id": 987,
      "section": "Section 26 — Coding & Algorithms for ML",
      "question": "Write code to detect and handle class imbalance via weighted sampling.",
      "difficulty": 1
    },
    {
      "id": 988,
      "section": "Section 26 — Coding & Algorithms for ML",
      "question": "Implement a basic attention mechanism (scaled dot-product) from scratch in NumPy/PyTorch.",
      "difficulty": 2
    },
    {
      "id": 989,
      "section": "Section 26 — Coding & Algorithms for ML",
      "question": "Write a function to tokenize text using a simple BPE-style merge algorithm.",
      "difficulty": 3
    },
    {
      "id": 990,
      "section": "Section 26 — Coding & Algorithms for ML",
      "question": "Implement top-k and top-p (nucleus) sampling from a probability distribution.",
      "difficulty": 2
    },
    {
      "id": 991,
      "section": "Section 26 — Coding & Algorithms for ML",
      "question": "Write a SQL query to compute rolling 7-day retention from an events table.",
      "difficulty": 3
    },
    {
      "id": 992,
      "section": "Section 26 — Coding & Algorithms for ML",
      "question": "Write a SQL query to detect duplicate near-matches in a customer table.",
      "difficulty": 2
    },
    {
      "id": 993,
      "section": "Section 26 — Coding & Algorithms for ML",
      "question": "Implement a basic LRU cache — relevant for semantic caching layers.",
      "difficulty": 1
    },
    {
      "id": 994,
      "section": "Section 26 — Coding & Algorithms for ML",
      "question": "Write code to chunk a long document into overlapping windows for embedding.",
      "difficulty": 3
    },
    {
      "id": 995,
      "section": "Section 26 — Coding & Algorithms for ML",
      "question": "Implement a simple priority queue-based approach to rank top-N recommendations efficiently.",
      "difficulty": 3
    },
    {
      "id": 996,
      "section": "Section 26 — Coding & Algorithms for ML",
      "question": "Write a function to batch API requests with retry/backoff for a rate-limited LLM endpoint.",
      "difficulty": 2
    },
    {
      "id": 997,
      "section": "Section 26 — Coding & Algorithms for ML",
      "question": "Implement a basic A/B test statistical significance calculator (two-proportion z-test).",
      "difficulty": 1
    },
    {
      "id": 998,
      "section": "Section 26 — Coding & Algorithms for ML",
      "question": "Write code to deduplicate embeddings above a similarity threshold efficiently.",
      "difficulty": 2
    },
    {
      "id": 999,
      "section": "Section 26 — Coding & Algorithms for ML",
      "question": "Implement gradient checking to validate a custom backpropagation implementation.",
      "difficulty": 2
    },
    {
      "id": 1000,
      "section": "Section 26 — Coding & Algorithms for ML",
      "question": "Write a function to parse and validate LLM JSON output against a schema, with error recovery.",
      "difficulty": 1
    },
    {
      "id": 1001,
      "section": "Section 26 — Coding & Algorithms for ML",
      "question": "Implement a circular buffer for maintaining a fixed-size sliding window of recent events.",
      "difficulty": 2
    },
    {
      "id": 1002,
      "section": "Section 26 — Coding & Algorithms for ML",
      "question": "Write code to compute exponential moving average for streaming metrics (e.g., drift detection).",
      "difficulty": 2
    },
    {
      "id": 1003,
      "section": "Section 26 — Coding & Algorithms for ML",
      "question": "Implement a basic beam search decoder from scratch.",
      "difficulty": 3
    },
    {
      "id": 1004,
      "section": "Section 26 — Coding & Algorithms for ML",
      "question": "Write a SQL query to compute cohort-based churn rate by signup month.",
      "difficulty": 2
    },
    {
      "id": 1005,
      "section": "Section 26 — Coding & Algorithms for ML",
      "question": "Implement reservoir sampling to maintain a random sample from a large/streaming dataset.",
      "difficulty": 3
    },
    {
      "id": 1006,
      "section": "Section 27 — Open-Ended Architecture Design Prompts",
      "question": "\"Our agent is slow and expensive — walk me through how you'd diagnose and fix it.\"",
      "difficulty": 3
    },
    {
      "id": 1007,
      "section": "Section 27 — Open-Ended Architecture Design Prompts",
      "question": "\"Design the AI architecture for a company going from 0 to 1 on GenAI features, with a 6-person team and a 6-month runway.\"",
      "difficulty": 3
    },
    {
      "id": 1008,
      "section": "Section 27 — Open-Ended Architecture Design Prompts",
      "question": "\"You inherit a RAG system with a 40% user-reported hallucination rate — what's your 30/60/90-day plan?\"",
      "difficulty": 3
    },
    {
      "id": 1009,
      "section": "Section 27 — Open-Ended Architecture Design Prompts",
      "question": "\"Design an AI platform that must serve both a consumer mobile app and an internal analyst tool with very different latency needs.\"",
      "difficulty": 3
    },
    {
      "id": 1010,
      "section": "Section 27 — Open-Ended Architecture Design Prompts",
      "question": "\"Your LLM provider just deprecated the model your production system depends on — walk through your response.\"",
      "difficulty": 3
    },
    {
      "id": 1011,
      "section": "Section 27 — Open-Ended Architecture Design Prompts",
      "question": "\"Design a system where cost per request must drop 70% in 3 months without a material quality drop — what levers do you pull, in what order?\"",
      "difficulty": 3
    },
    {
      "id": 1012,
      "section": "Section 27 — Open-Ended Architecture Design Prompts",
      "question": "\"You're asked to add an AI feature to a HIPAA-regulated product — how does that change your architecture from the ground up?\"",
      "difficulty": 3
    },
    {
      "id": 1013,
      "section": "Section 27 — Open-Ended Architecture Design Prompts",
      "question": "\"Design an architecture that supports both a fast-moving experimental team and a stability-critical production team sharing the same model infrastructure.\"",
      "difficulty": 3
    },
    {
      "id": 1014,
      "section": "Section 27 — Open-Ended Architecture Design Prompts",
      "question": "\"Your evaluation metrics show a model is 'better,' but a key customer says quality dropped — how do you reconcile this?\"",
      "difficulty": 3
    },
    {
      "id": 1015,
      "section": "Section 27 — Open-Ended Architecture Design Prompts",
      "question": "\"Design a system to let 200 internal teams build AI features without each team reinventing prompt management, evals, and guardrails.\"",
      "difficulty": 3
    },
    {
      "id": 1016,
      "section": "Section 27 — Open-Ended Architecture Design Prompts",
      "question": "\"You must choose between a $2M/year managed AI platform and a 4-engineer team building in-house — walk through your decision framework.\"",
      "difficulty": 3
    },
    {
      "id": 1017,
      "section": "Section 27 — Open-Ended Architecture Design Prompts",
      "question": "\"Design the rollback and incident-response plan for an AI feature that starts giving financial advice it shouldn't.\"",
      "difficulty": 3
    },
    {
      "id": 1018,
      "section": "Section 27 — Open-Ended Architecture Design Prompts",
      "question": "\"How would you architect a system to detect, within minutes, that a newly deployed prompt change has made outputs worse?\"",
      "difficulty": 3
    },
    {
      "id": 1019,
      "section": "Section 27 — Open-Ended Architecture Design Prompts",
      "question": "\"Design an AI architecture resilient to a single point of failure at every layer — model, retrieval, infra, data.\"",
      "difficulty": 3
    },
    {
      "id": 1020,
      "section": "Section 27 — Open-Ended Architecture Design Prompts",
      "question": "\"Your fastest-growing product feature is an LLM agent, but its cost is growing faster than revenue — what do you do?\"",
      "difficulty": 3
    },
    {
      "id": 1021,
      "section": "Section 27 — Open-Ended Architecture Design Prompts",
      "question": "\"Design a system where three different business units want to use three different LLM providers — how do you standardize without forcing lock-step migration?\"",
      "difficulty": 3
    },
    {
      "id": 1022,
      "section": "Section 27 — Open-Ended Architecture Design Prompts",
      "question": "\"You need to demonstrate AI ROI to the board in 90 days — what would you build/measure first?\"",
      "difficulty": 3
    },
    {
      "id": 1023,
      "section": "Section 27 — Open-Ended Architecture Design Prompts",
      "question": "\"Design an architecture for a global product needing consistent AI quality across 15 languages with very different available training data.\"",
      "difficulty": 3
    },
    {
      "id": 1024,
      "section": "Section 27 — Open-Ended Architecture Design Prompts",
      "question": "\"How would you structure a 'model risk committee' review for a new high-stakes AI feature, and what artifacts would you bring?\"",
      "difficulty": 3
    },
    {
      "id": 1025,
      "section": "Section 27 — Open-Ended Architecture Design Prompts",
      "question": "\"Design a fallback architecture so that if every LLM provider is down simultaneously, the product still functions in a degraded mode.\"",
      "difficulty": 3
    },
    {
      "id": 1026,
      "section": "Section 27 — Open-Ended Architecture Design Prompts",
      "question": "\"Your team wants to fine-tune a model; a rival team says RAG is enough — how do you settle this with evidence, not opinion?\"",
      "difficulty": 3
    },
    {
      "id": 1027,
      "section": "Section 27 — Open-Ended Architecture Design Prompts",
      "question": "\"Design an AI system architecture where a single bad actor must not be able to cause more than $X of damage even with full access.\"",
      "difficulty": 3
    },
    {
      "id": 1028,
      "section": "Section 27 — Open-Ended Architecture Design Prompts",
      "question": "\"How would you architect a system where the model itself is a fast-moving research artifact but the surrounding product must be rock-solid?\"",
      "difficulty": 3
    },
    {
      "id": 1029,
      "section": "Section 27 — Open-Ended Architecture Design Prompts",
      "question": "\"Design an evaluation and rollout process that lets you ship a new model to production within 24 hours of a provider release, safely.\"",
      "difficulty": 3
    },
    {
      "id": 1030,
      "section": "Section 27 — Open-Ended Architecture Design Prompts",
      "question": "\"You discover your training data includes a substantial amount of low-quality scraped content — what's your remediation plan?\"",
      "difficulty": 3
    },
    {
      "id": 1031,
      "section": "Section 27 — Open-Ended Architecture Design Prompts",
      "question": "\"Design the architecture and governance for an AI feature that will make automated decisions with legal consequences for users.\"",
      "difficulty": 3
    },
    {
      "id": 1032,
      "section": "Section 27 — Open-Ended Architecture Design Prompts",
      "question": "\"How would you design your AI platform's roadmap knowing frontier model capabilities will meaningfully change every 6 months?\"",
      "difficulty": 3
    },
    {
      "id": 1033,
      "section": "Section 27 — Open-Ended Architecture Design Prompts",
      "question": "\"Design a system to let product managers self-serve simple AI features without engineering involvement, safely.\"",
      "difficulty": 3
    },
    {
      "id": 1034,
      "section": "Section 27 — Open-Ended Architecture Design Prompts",
      "question": "\"Your AI system passed every offline eval but failed publicly on launch day — walk through your root-cause process.\"",
      "difficulty": 3
    },
    {
      "id": 1035,
      "section": "Section 27 — Open-Ended Architecture Design Prompts",
      "question": "\"Design the long-term (3-year) architecture for an AI platform assuming inference cost drops 10x but data/governance requirements double.\"",
      "difficulty": 3
    },
    {
      "id": 1036,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "Why does KL divergence show up in RLHF/DPO objectives?",
      "difficulty": 2
    },
    {
      "id": 1037,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "Compare greedy decoding, beam search, and nucleus (top-p) sampling.",
      "difficulty": 3
    },
    {
      "id": 1038,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "What are common distributed-training failure modes (stragglers, gradient explosion, checkpoint corruption)?",
      "difficulty": 2
    },
    {
      "id": 1039,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "Compare diffusion models vs. autoregressive generation for image/multimodal tasks.",
      "difficulty": 3
    },
    {
      "id": 1040,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "What breaks first when you push context length far beyond a model's training distribution?",
      "difficulty": 3
    },
    {
      "id": 1041,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "When would prompt engineering alone fail and force you toward fine-tuning?",
      "difficulty": 3
    },
    {
      "id": 1042,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "Why does batch size interact with learning rate, and how do you scale one when changing the other?",
      "difficulty": 3
    },
    {
      "id": 1043,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "What's the practical difference between fine-tuning the full model vs. just the last few layers?",
      "difficulty": 3
    },
    {
      "id": 1044,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "Why do larger models sometimes hallucinate less, and sometimes more confidently, than smaller ones?",
      "difficulty": 3
    },
    {
      "id": 1045,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "What's the effect of temperature=0 on reproducibility, and why isn't it perfectly deterministic in practice?",
      "difficulty": 2
    },
    {
      "id": 1046,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "Why does RAG sometimes make hallucination worse instead of better?",
      "difficulty": 2
    },
    {
      "id": 1047,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "What's the tradeoff of adding more retrieved chunks to a prompt beyond a certain point?",
      "difficulty": 3
    },
    {
      "id": 1048,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "Why do embedding models trained on one domain often underperform on another without fine-tuning?",
      "difficulty": 3
    },
    {
      "id": 1049,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "What causes a model to ignore instructions buried in the middle of a long system prompt?",
      "difficulty": 2
    },
    {
      "id": 1050,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "Why does increasing model size not always improve reasoning tasks proportionally to language tasks?",
      "difficulty": 3
    },
    {
      "id": 1051,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "What's the difference between a model being \"aligned\" and a model being \"safe\"?",
      "difficulty": 3
    },
    {
      "id": 1052,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "Why might two models with identical benchmark scores behave very differently on your specific use case?",
      "difficulty": 3
    },
    {
      "id": 1053,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "What causes cost estimates for an LLM feature to be wildly wrong in production vs. testing?",
      "difficulty": 2
    },
    {
      "id": 1054,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "Why does adding more agents to a multi-agent system sometimes reduce overall task success rate?",
      "difficulty": 3
    },
    {
      "id": 1055,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "What's the practical failure mode of over-relying on LLM-as-judge for evaluation?",
      "difficulty": 3
    },
    {
      "id": 1056,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "Why do smaller, well-tuned models sometimes outperform larger general-purpose ones on narrow tasks?",
      "difficulty": 2
    },
    {
      "id": 1057,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "What causes latency variance (not just average latency) to spike under production load for LLM serving?",
      "difficulty": 3
    },
    {
      "id": 1058,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "Why does streaming output change your error-handling design compared to non-streaming responses?",
      "difficulty": 2
    },
    {
      "id": 1059,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "What's the risk of caching LLM responses too aggressively for a personalized product?",
      "difficulty": 3
    },
    {
      "id": 1060,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "Why can a model pass all unit-test-style evals but still fail in real conversations?",
      "difficulty": 3
    },
    {
      "id": 1061,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "What causes token-count estimates to diverge from actual billed tokens across providers?",
      "difficulty": 3
    },
    {
      "id": 1062,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "Why does fine-tuning sometimes reduce a model's general capability even on unrelated tasks?",
      "difficulty": 3
    },
    {
      "id": 1063,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "What's the failure mode of a guardrail system that's too aggressive vs. too permissive?",
      "difficulty": 3
    },
    {
      "id": 1064,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "Why does a RAG system's quality often degrade after a document-format change upstream, silently?",
      "difficulty": 2
    },
    {
      "id": 1065,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "What causes vector search recall to drop as an index grows, even with the same algorithm?",
      "difficulty": 2
    },
    {
      "id": 1066,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "Why is p99 latency often a better SLA target than average latency for an LLM API?",
      "difficulty": 2
    },
    {
      "id": 1067,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "What's the risk of an agent's tool schema being too generic vs. too specific?",
      "difficulty": 2
    },
    {
      "id": 1068,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "Why does model behavior sometimes change after a provider's \"silent\" backend update with no version bump?",
      "difficulty": 3
    },
    {
      "id": 1069,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "What causes a well-performing offline eval to fail to predict real user satisfaction?",
      "difficulty": 3
    },
    {
      "id": 1070,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "Why does context compression sometimes lose exactly the detail that mattered for the final answer?",
      "difficulty": 2
    },
    {
      "id": 1071,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "What's the tradeoff of using a single mega-prompt vs. decomposing into multiple smaller LLM calls?",
      "difficulty": 3
    },
    {
      "id": 1072,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "Why can increasing few-shot examples past a certain point hurt performance instead of helping?",
      "difficulty": 3
    },
    {
      "id": 1073,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "What causes cost per query to blow up quietly when an agent enters a retry loop?",
      "difficulty": 3
    },
    {
      "id": 1074,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "Why is \"the model said so\" an insufficient explanation for a production incident review?",
      "difficulty": 2
    },
    {
      "id": 1075,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "What's the risk of conflating model capability improvements with actual product-quality improvements?",
      "difficulty": 2
    },
    {
      "id": 1076,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "Why does data drift sometimes matter more for feature pipelines than for the model itself?",
      "difficulty": 3
    },
    {
      "id": 1077,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "What causes two teams' \"same\" eval scores to be non-comparable across different eval harness implementations?",
      "difficulty": 2
    },
    {
      "id": 1078,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "Why might reducing hallucination rate not actually improve user trust metrics?",
      "difficulty": 3
    },
    {
      "id": 1079,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "What's the failure mode of over-indexing on one benchmark when selecting a foundation model?",
      "difficulty": 2
    },
    {
      "id": 1080,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "Why does model quantization sometimes disproportionately hurt performance on non-English languages?",
      "difficulty": 3
    },
    {
      "id": 1081,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "What causes an LLM system to behave inconsistently across identical repeated requests even at low temperature?",
      "difficulty": 3
    },
    {
      "id": 1082,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "Why is \"add more guardrails\" often the wrong first response to a safety incident?",
      "difficulty": 3
    },
    {
      "id": 1083,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "What's the risk of building critical business logic entirely inside a prompt rather than in code?",
      "difficulty": 3
    },
    {
      "id": 1084,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "Why does an agent's plan sometimes look correct step-by-step but fail to achieve the actual goal?",
      "difficulty": 3
    },
    {
      "id": 1085,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "What causes retrieval-augmented answers to cite the wrong source even when the right one was retrieved?",
      "difficulty": 3
    },
    {
      "id": 1086,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "Why might a smaller context window sometimes produce more reliable output than a larger one?",
      "difficulty": 3
    },
    {
      "id": 1087,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "What's the practical limit of chain-of-thought prompting's benefit as task complexity increases?",
      "difficulty": 3
    },
    {
      "id": 1088,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "Why does model choice interact with prompt design — i.e., why isn't a \"good prompt\" portable across models?",
      "difficulty": 3
    },
    {
      "id": 1089,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "What causes teams to underestimate the ongoing maintenance cost of an LLM feature after initial launch?",
      "difficulty": 3
    },
    {
      "id": 1090,
      "section": "Section 28 — Rapid-Fire Depth Probes",
      "question": "Why is \"it works in the demo\" one of the least reliable signals of production readiness for an AI system?",
      "difficulty": 3
    },
    {
      "id": 1091,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "Explain the NIST AI Risk Management Framework's four core functions (Govern, Map, Measure, Manage) and how you'd operationalize each at an enterprise.",
      "difficulty": 3
    },
    {
      "id": 1092,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "What is ISO/IEC 42001, and how does an AI Management System (AIMS) certification differ from a one-off compliance checklist?",
      "difficulty": 3
    },
    {
      "id": 1093,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "Design a 5-level AI maturity model for an enterprise (from ad hoc experimentation to fully governed, optimized AI operations) and define what distinguishes each level.",
      "difficulty": 3
    },
    {
      "id": 1094,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "Build a total cost of ownership (TCO) framework for an enterprise AI system — what cost categories are commonly underestimated?",
      "difficulty": 3
    },
    {
      "id": 1095,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "Design a build-vs-buy scoring methodology (weighted scorecard) for evaluating a new AI capability.",
      "difficulty": 3
    },
    {
      "id": 1096,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "Design a vendor evaluation scorecard for selecting an enterprise LLM/AI platform provider — what dimensions matter beyond price and benchmark scores?",
      "difficulty": 3
    },
    {
      "id": 1097,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "Compare Databricks, Snowflake Cortex, Palantir AIP, and Microsoft Fabric as enterprise AI/data platforms — when would you choose each?",
      "difficulty": 3
    },
    {
      "id": 1098,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "How would you integrate an LLM-powered feature into an existing SAP ERP environment without disrupting core transactional systems?",
      "difficulty": 3
    },
    {
      "id": 1099,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "Design a pattern for embedding AI capabilities into Salesforce (e.g., Einstein-style) without creating a shadow-IT parallel system.",
      "difficulty": 3
    },
    {
      "id": 1100,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "How would you integrate a GenAI assistant into ServiceNow for IT service management use cases?",
      "difficulty": 3
    },
    {
      "id": 1101,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "Design a RACI matrix for an enterprise AI Center of Excellence spanning legal, security, data engineering, ML platform, and product teams.",
      "difficulty": 3
    },
    {
      "id": 1102,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "What is a federated AI operating model, and how does it differ from a centralized AI CoE at enterprise scale?",
      "difficulty": 3
    },
    {
      "id": 1103,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "Design an executive/board-level one-pager template for communicating an AI initiative's status, risk, and ROI.",
      "difficulty": 3
    },
    {
      "id": 1104,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "How would you structure a change-management program for AI adoption across a 5,000-person enterprise resistant to workflow changes?",
      "difficulty": 3
    },
    {
      "id": 1105,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "Design a migration plan moving a legacy rules-based enterprise system to an AI-augmented architecture without a \"big bang\" cutover.",
      "difficulty": 3
    },
    {
      "id": 1106,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "How would you architect multi-modal enterprise data integration combining structured ERP data, unstructured documents, and image/scan data into one AI-accessible layer?",
      "difficulty": 3
    },
    {
      "id": 1107,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "What contract/procurement terms should legal specifically negotiate with an enterprise AI vendor (SLAs, data processing agreements, indemnification, model-deprecation notice periods)?",
      "difficulty": 3
    },
    {
      "id": 1108,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "Design an AI initiative portfolio management framework for a CIO/CTO managing 30+ concurrent AI projects across business units.",
      "difficulty": 3
    },
    {
      "id": 1109,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "What are the core responsibilities of a Chief AI Officer role, and how does it differ from a VP of Engineering or Chief Data Officer?",
      "difficulty": 3
    },
    {
      "id": 1110,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "Design an enterprise-wide prompt and knowledge-asset governance system — how do you prevent 50 teams from creating 50 inconsistent, redundant prompt libraries?",
      "difficulty": 3
    },
    {
      "id": 1111,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "Explain the FDA's regulatory framework for AI/ML-based Software as a Medical Device (SaMD), and how a \"locked\" vs \"adaptive\" algorithm changes compliance requirements.",
      "difficulty": 3
    },
    {
      "id": 1112,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "What is the NAIC's model governance guidance for AI in insurance underwriting, and how does it compare to SR 11-7 in banking?",
      "difficulty": 3
    },
    {
      "id": 1113,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "Design an enterprise data classification scheme (public/internal/confidential/restricted) and show how it should gate what data can flow to which AI systems.",
      "difficulty": 3
    },
    {
      "id": 1114,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "How would you architect a \"walled garden\" AI environment for a highly regulated enterprise (defense, pharma) where no data can leave a controlled boundary, including for model updates?",
      "difficulty": 3
    },
    {
      "id": 1115,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "Design an enterprise-wide AI incident severity classification (SEV1-SEV4 equivalent) and the corresponding response SLA for each tier.",
      "difficulty": 3
    },
    {
      "id": 1116,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "How would you structure quarterly AI governance reporting to a board risk committee?",
      "difficulty": 3
    },
    {
      "id": 1117,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "What is shadow AI (unsanctioned tool usage by employees), and how would you design a policy and technical control response to it?",
      "difficulty": 3
    },
    {
      "id": 1118,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "Design an enterprise single sign-on and entitlement model for AI tools ensuring an employee's AI access mirrors their existing data access rights exactly.",
      "difficulty": 3
    },
    {
      "id": 1119,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "How would you build a business case comparing the TCO of a single enterprise-wide AI platform versus allowing each business unit to independently license tools?",
      "difficulty": 3
    },
    {
      "id": 1120,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "What KPIs would you present to a CFO to justify continued AI platform investment after the first year, beyond raw usage numbers?",
      "difficulty": 3
    },
    {
      "id": 1121,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "Design an AI procurement due-diligence checklist covering model provenance, training-data licensing, and downstream liability exposure.",
      "difficulty": 3
    },
    {
      "id": 1122,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "How would you structure an AI ethics review board's charter, including escalation authority and how it differs from a technical architecture review board?",
      "difficulty": 3
    },
    {
      "id": 1123,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "What is the EU AI Act's \"high-risk\" system obligations (conformity assessment, technical documentation, human oversight) and how would you build a compliance-readiness checklist against them?",
      "difficulty": 3
    },
    {
      "id": 1124,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "Design an enterprise AI skills/capability matrix used for both hiring and internal upskilling planning across an engineering organization.",
      "difficulty": 3
    },
    {
      "id": 1125,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "How would you present a \"walk before you run\" AI adoption sequence to a board that wants to move directly to autonomous agents?",
      "difficulty": 3
    },
    {
      "id": 1126,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "What is vendor lock-in risk specific to enterprise AI platforms, and how would you structure contracts/architecture to preserve exit optionality?",
      "difficulty": 3
    },
    {
      "id": 1127,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "Design a cross-business-unit AI use-case intake and prioritization committee process for a large enterprise.",
      "difficulty": 3
    },
    {
      "id": 1128,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "How would you calculate and present the \"cost of inaction\" — the competitive risk of not investing in an AI capability — to a skeptical executive team?",
      "difficulty": 3
    },
    {
      "id": 1129,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "What due diligence would you perform before allowing an AI vendor's model to process data subject to attorney-client privilege?",
      "difficulty": 3
    },
    {
      "id": 1130,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "Design an enterprise data residency and sovereign-cloud architecture for a company operating in the EU, US, China, and India simultaneously.",
      "difficulty": 3
    },
    {
      "id": 1131,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "How would you architect AI system access for third-party contractors/consultants without granting them the same data visibility as full-time employees?",
      "difficulty": 3
    },
    {
      "id": 1132,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "What is a model transparency/nutrition-label approach to enterprise AI procurement, and what should it disclose?",
      "difficulty": 3
    },
    {
      "id": 1133,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "Design a business continuity plan specifically for AI-dependent enterprise workflows if the AI platform team is unavailable (turnover, reorg) for an extended period.",
      "difficulty": 3
    },
    {
      "id": 1134,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "How would you structure an internal AI \"marketplace\" where business units can discover and request access to vetted, pre-approved AI capabilities?",
      "difficulty": 3
    },
    {
      "id": 1135,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "What enterprise architecture principles (from a TOGAF-style framework) apply most directly to governing AI system sprawl?",
      "difficulty": 3
    },
    {
      "id": 1136,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "How would you design an AI capability's decommissioning/sunset process at enterprise scale, including data retention and dependent-system notification?",
      "difficulty": 3
    },
    {
      "id": 1137,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "Design a cross-functional incident command structure specifically for a major AI-driven outage affecting multiple business units simultaneously.",
      "difficulty": 3
    },
    {
      "id": 1138,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "What's the enterprise-grade difference between a proof-of-concept, a pilot, and a production-grade AI deployment, and what gate criteria separate each stage?",
      "difficulty": 3
    },
    {
      "id": 1139,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "How would you structure an annual AI risk assessment cycle that satisfies both internal audit and external regulatory expectations?",
      "difficulty": 3
    },
    {
      "id": 1140,
      "section": "Section 29 — Enterprise AI Governance, Frameworks, Platforms & Executive Communication",
      "question": "Design a framework for measuring and reporting AI-driven productivity gains at the enterprise level without over-claiming causality.",
      "difficulty": 3
    },
    {
      "id": 1141,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "Explain the Model Context Protocol (MCP): what problem does it solve, and how does it differ from a custom tool-calling integration?",
      "difficulty": 3
    },
    {
      "id": 1142,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "What is an MCP server vs. an MCP client, and how does the client-server architecture map onto an enterprise's existing systems?",
      "difficulty": 3
    },
    {
      "id": 1143,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "Explain the MCP Registry concept — how is it analogous to a package registry like Docker Hub or npm, and what enterprise problem does it solve?",
      "difficulty": 3
    },
    {
      "id": 1144,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "What is an \"MCP Server Card,\" and how does it enable discovery without a live connection?",
      "difficulty": 3
    },
    {
      "id": 1145,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "How would you design governance for an internal MCP server registry — namespace trust, pre-audit requirements, and versioning?",
      "difficulty": 3
    },
    {
      "id": 1146,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "Explain MCP's shift toward a stateless architecture — why does statefulness cause problems at enterprise scale, and what does stateless enable?",
      "difficulty": 3
    },
    {
      "id": 1147,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "What is the MCP \"Tasks\" extension, and why does it matter for long-running agent operations?",
      "difficulty": 3
    },
    {
      "id": 1148,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "How would you design authentication/authorization for MCP servers at enterprise scale, given the protocol's move toward OAuth/OpenID Connect alignment?",
      "difficulty": 3
    },
    {
      "id": 1149,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "What is \"Elicitation\" in MCP, and how does it enable human-in-the-loop approval for high-risk agent actions?",
      "difficulty": 3
    },
    {
      "id": 1150,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "Design an enterprise MCP gateway that sits between internal agents and a mix of internal and third-party MCP servers — what does it need to enforce?",
      "difficulty": 3
    },
    {
      "id": 1151,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "Explain the Agent2Agent (A2A) protocol: what problem does it solve that MCP does not?",
      "difficulty": 3
    },
    {
      "id": 1152,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "What is an \"Agent Card\" in A2A, and how does it enable one agent to discover another agent's capabilities across organizational boundaries?",
      "difficulty": 3
    },
    {
      "id": 1153,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "Explain A2A's task lifecycle states (submitted, working, input-required, completed, failed, canceled, rejected) and why an explicit lifecycle matters for enterprise workflows.",
      "difficulty": 3
    },
    {
      "id": 1154,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "How do MCP and A2A compose together in a single enterprise architecture — which layer handles what?",
      "difficulty": 3
    },
    {
      "id": 1155,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "Compare A2A, MCP, ACP (IBM's Agent Communication Protocol), and ANP (Agent Network Protocol) — what distinct problem does each address?",
      "difficulty": 3
    },
    {
      "id": 1156,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "Design an enterprise Agent Registry — what should it catalog beyond just an agent's name (capabilities, owner, risk tier, data access scope)?",
      "difficulty": 3
    },
    {
      "id": 1157,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "What is an \"Agent Broker,\" and how does it differ architecturally from an Agent Registry?",
      "difficulty": 3
    },
    {
      "id": 1158,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "How would you design secure data hand-off between two agents built by different teams (e.g., a Sales agent passing context to a Pricing agent) without redundant re-querying or data leakage?",
      "difficulty": 3
    },
    {
      "id": 1159,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "Design a governance review process specifically for onboarding a new agent into an enterprise Agent Registry before it's discoverable by other agents.",
      "difficulty": 3
    },
    {
      "id": 1160,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "What security risks are introduced specifically by cross-vendor agent interoperability (A2A-style) that don't exist in a single-vendor, single-agent system?",
      "difficulty": 3
    },
    {
      "id": 1161,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "How would you audit and trace a multi-agent workflow that spans agents from three different vendors communicating via A2A, when something goes wrong?",
      "difficulty": 3
    },
    {
      "id": 1162,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "What is the \"governance gap\" in current agent interoperability protocols — what can MCP, A2A, and ACP not yet express natively that enterprises need?",
      "difficulty": 3
    },
    {
      "id": 1163,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "Design a permission model for an agent that uses MCP to access ten different internal tools with very different sensitivity levels.",
      "difficulty": 3
    },
    {
      "id": 1164,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "How would you version and deprecate an internal MCP server without breaking every agent currently depending on it?",
      "difficulty": 3
    },
    {
      "id": 1165,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "What observability specifically changes when your agent architecture spans MCP (tool access) and A2A (agent-to-agent) simultaneously?",
      "difficulty": 3
    },
    {
      "id": 1166,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "Design a \"walled garden\" MCP deployment for a regulated enterprise that cannot allow agents to reach external MCP servers at all.",
      "difficulty": 3
    },
    {
      "id": 1167,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "How would you decide whether a new integration should be built as an MCP server, an A2A-exposed agent, or a traditional internal API?",
      "difficulty": 3
    },
    {
      "id": 1168,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "What is permission-aware retrieval in enterprise RAG, and why do most consumer-grade RAG tools fail to provide it?",
      "difficulty": 3
    },
    {
      "id": 1169,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "Design a RAG system that automatically inherits a document's existing access-control permissions rather than requiring separate, manually-maintained AI permissions.",
      "difficulty": 3
    },
    {
      "id": 1170,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "Explain hybrid RAG, GraphRAG, and Agentic RAG as a maturity progression — when does each level of complexity actually earn its cost in an enterprise context?",
      "difficulty": 3
    },
    {
      "id": 1171,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "What is late chunking, and how does it solve a specific failure mode of the traditional chunk-then-embed pipeline?",
      "difficulty": 3
    },
    {
      "id": 1172,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "Design an audit-trail/lineage system for enterprise RAG that can trace any generated answer back to its exact source document and permission context, for regulatory defensibility.",
      "difficulty": 3
    },
    {
      "id": 1173,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "What is the \"facts vs. behavior\" heuristic for choosing between RAG and fine-tuning, and where does it break down?",
      "difficulty": 3
    },
    {
      "id": 1174,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "How would you design a RAG evaluation framework with measurable, auditable thresholds suitable for a regulated enterprise (not just an internal quality bar)?",
      "difficulty": 3
    },
    {
      "id": 1175,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "What is shadow AI risk specific to RAG systems, and how does uncoordinated departmental RAG deployment create compliance exposure?",
      "difficulty": 3
    },
    {
      "id": 1176,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "Design an enterprise knowledge layer that unifies RAG-based retrieval across multiple AI interfaces (chatbot, IDE assistant, CRM agent) with one consistent permission and audit system.",
      "difficulty": 3
    },
    {
      "id": 1177,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "How would you decide when single-pass RAG is insufficient and agentic/corrective RAG (re-searching on insufficient evidence) is actually warranted, given the added cost and latency?",
      "difficulty": 3
    },
    {
      "id": 1178,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "What embedding-model and re-ranker landscape considerations matter for an enterprise choosing a RAG stack in 2026 versus building on defaults from a year or two prior?",
      "difficulty": 3
    },
    {
      "id": 1179,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "Design a self-improving RAG system where verification workflows and expert feedback propagate corrections automatically across all connected interfaces.",
      "difficulty": 3
    },
    {
      "id": 1180,
      "section": "Section 30 — Enterprise Agent Interoperability (MCP, A2A) & Advanced RAG",
      "question": "How would you present a board-level risk assessment of your organization's agent interoperability posture (MCP/A2A exposure, third-party agent access, registry governance maturity)?",
      "difficulty": 3
    },
    {
      "id": 1181,
      "section": "Section 31 — Cloud-Native Agent Deployment: AWS, Azure, GCP",
      "question": "Compare AWS Bedrock AgentCore, Azure AI Foundry Agent Service, and Google Vertex AI Agent Engine as managed agent runtimes — what does \"managed runtime\" actually need to provide beyond model access?",
      "difficulty": 3
    },
    {
      "id": 1182,
      "section": "Section 31 — Cloud-Native Agent Deployment: AWS, Azure, GCP",
      "question": "Explain AWS Bedrock's Action Groups pattern — how does an agent get tool access, and what AWS service actually executes each action?",
      "difficulty": 3
    },
    {
      "id": 1183,
      "section": "Section 31 — Cloud-Native Agent Deployment: AWS, Azure, GCP",
      "question": "What is AgentCore's approach to identity and token management, and why is it described as well-suited for zero-trust, multi-tenant deployments?",
      "difficulty": 3
    },
    {
      "id": 1184,
      "section": "Section 31 — Cloud-Native Agent Deployment: AWS, Azure, GCP",
      "question": "How does Azure AI Foundry's agent identity model work, and what's the tradeoff for enterprises not already using Microsoft Entra ID?",
      "difficulty": 3
    },
    {
      "id": 1185,
      "section": "Section 31 — Cloud-Native Agent Deployment: AWS, Azure, GCP",
      "question": "Explain how Google Vertex AI Agent Engine handles identity and IAM permissions for deployed agents, and how Apigee fits into the architecture.",
      "difficulty": 3
    },
    {
      "id": 1186,
      "section": "Section 31 — Cloud-Native Agent Deployment: AWS, Azure, GCP",
      "question": "What is Bedrock AgentCore's approach to session memory, and what underlying AWS service backs it?",
      "difficulty": 3
    },
    {
      "id": 1187,
      "section": "Section 31 — Cloud-Native Agent Deployment: AWS, Azure, GCP",
      "question": "Compare the observability approach across all three platforms — CloudWatch tracing (AWS), Azure Monitor integration, and Vertex AI's built-in dashboards.",
      "difficulty": 3
    },
    {
      "id": 1188,
      "section": "Section 31 — Cloud-Native Agent Deployment: AWS, Azure, GCP",
      "question": "Design a multi-agent collaboration architecture using Bedrock AgentCore's 2026 multi-agent delegation capability — how do sub-agents get invoked?",
      "difficulty": 3
    },
    {
      "id": 1189,
      "section": "Section 31 — Cloud-Native Agent Deployment: AWS, Azure, GCP",
      "question": "Why would an enterprise choose Azure AI Foundry specifically for GPT-5/OpenAI-model-based agents, given model availability differences across the three clouds?",
      "difficulty": 3
    },
    {
      "id": 1190,
      "section": "Section 31 — Cloud-Native Agent Deployment: AWS, Azure, GCP",
      "question": "What does deep Microsoft 365 integration (Outlook, Teams, SharePoint, Sentinel) unlock for an Azure-deployed agent that AWS/GCP-deployed agents can't easily replicate?",
      "difficulty": 3
    },
    {
      "id": 1191,
      "section": "Section 31 — Cloud-Native Agent Deployment: AWS, Azure, GCP",
      "question": "How does Vertex AI's Google Search grounding with citation support change the RAG-vs-native-search tradeoff for a GCP-native agent?",
      "difficulty": 3
    },
    {
      "id": 1192,
      "section": "Section 31 — Cloud-Native Agent Deployment: AWS, Azure, GCP",
      "question": "Design a decision framework for choosing between AWS Bedrock, Azure AI Foundry, and Vertex AI for a new enterprise agent deployment, given existing cloud investment.",
      "difficulty": 3
    },
    {
      "id": 1193,
      "section": "Section 31 — Cloud-Native Agent Deployment: AWS, Azure, GCP",
      "question": "What compliance/certification differences exist across the three platforms (FedRAMP, HIPAA, data residency), and how would this affect a regulated-industry deployment decision?",
      "difficulty": 3
    },
    {
      "id": 1194,
      "section": "Section 31 — Cloud-Native Agent Deployment: AWS, Azure, GCP",
      "question": "How would you architect a multi-cloud agent deployment needing GPT, Claude, and Llama all under enterprise terms — why do practitioners suggest this typically requires two platforms, not one?",
      "difficulty": 3
    },
    {
      "id": 1195,
      "section": "Section 31 — Cloud-Native Agent Deployment: AWS, Azure, GCP",
      "question": "Explain how transforming existing internal APIs into MCP servers via Apigee (GCP) compares to building MCP servers from scratch — what governance advantage does this pattern offer?",
      "difficulty": 3
    },
    {
      "id": 1196,
      "section": "Section 31 — Cloud-Native Agent Deployment: AWS, Azure, GCP",
      "question": "Design cost controls for an agent platform processing millions of sessions monthly across a hyperscaler's consumption-based pricing model — what usage patterns most commonly cause runaway cost?",
      "difficulty": 3
    },
    {
      "id": 1197,
      "section": "Section 31 — Cloud-Native Agent Deployment: AWS, Azure, GCP",
      "question": "What is VPC/PrivateLink-based network isolation for agent deployments, and why does AWS's implementation get specifically called out as strong for regulated environments?",
      "difficulty": 3
    },
    {
      "id": 1198,
      "section": "Section 31 — Cloud-Native Agent Deployment: AWS, Azure, GCP",
      "question": "How would you design an agent's tool-execution layer to be portable across AWS Lambda (Bedrock Action Groups), Azure Functions (Foundry), and GCP Cloud Functions (Vertex) without vendor-locking the core agent logic?",
      "difficulty": 3
    },
    {
      "id": 1199,
      "section": "Section 31 — Cloud-Native Agent Deployment: AWS, Azure, GCP",
      "question": "What governance gap does the OutSystems 2026 finding (96% of enterprises have agents in production, only 12% can govern them) point to, and how would you close it architecturally regardless of which cloud you're on?",
      "difficulty": 3
    },
    {
      "id": 1200,
      "section": "Section 31 — Cloud-Native Agent Deployment: AWS, Azure, GCP",
      "question": "Design an evaluation/policy-preview integration (like Bedrock AgentCore's Policy and Evaluations previews) into a CI/CD pipeline for agent deployment.",
      "difficulty": 3
    },
    {
      "id": 1201,
      "section": "Section 31 — Cloud-Native Agent Deployment: AWS, Azure, GCP",
      "question": "How would you decide whether to build on a hyperscaler's native agent runtime versus an open-source framework (LangGraph, CrewAI) versus a cross-cloud platform — what does each trade away?",
      "difficulty": 3
    },
    {
      "id": 1202,
      "section": "Section 31 — Cloud-Native Agent Deployment: AWS, Azure, GCP",
      "question": "What does \"agent estate\" mean as an enterprise planning concept, and how would you inventory and govern one across multiple cloud deployments?",
      "difficulty": 3
    },
    {
      "id": 1203,
      "section": "Section 31 — Cloud-Native Agent Deployment: AWS, Azure, GCP",
      "question": "Design disaster recovery for a mission-critical agent deployed on a single hyperscaler's managed runtime — what's actually portable if that cloud has an extended outage?",
      "difficulty": 3
    },
    {
      "id": 1204,
      "section": "Section 31 — Cloud-Native Agent Deployment: AWS, Azure, GCP",
      "question": "How would you structure IAM least-privilege permissions for an agent on Vertex AI Agent Engine that needs to query BigQuery but never modify it?",
      "difficulty": 3
    },
    {
      "id": 1205,
      "section": "Section 31 — Cloud-Native Agent Deployment: AWS, Azure, GCP",
      "question": "What is the practical difference between \"agent orchestration inside one cloud's estate\" (e.g., Bedrock Agents and Flows) versus true cross-cloud agent portability, and which do most enterprises actually need?",
      "difficulty": 3
    },
    {
      "id": 1206,
      "section": "Section 31 — Cloud-Native Agent Deployment: AWS, Azure, GCP",
      "question": "How would you present a cloud-agent-platform selection recommendation to a CTO who wants to avoid a repeat of a prior costly cloud-migration lock-in mistake?",
      "difficulty": 3
    },
    {
      "id": 1207,
      "section": "Section 32 — Multimodal AI & Vision-Language Models",
      "question": "How do early fusion and late fusion architectures compare when designing a vision-language model for dense video captioning?",
      "difficulty": 2
    },
    {
      "id": 1208,
      "section": "Section 32 — Multimodal AI & Vision-Language Models",
      "question": "Why have simple MLP projectors largely replaced complex Q-Formers in recent multimodal architectures like LLaVA-1.5?",
      "difficulty": 3
    },
    {
      "id": 1209,
      "section": "Section 32 — Multimodal AI & Vision-Language Models",
      "question": "Contrast the embedding spaces of CLIP and SigLIP; in what specific data scenarios does SigLIP's pairwise sigmoid loss outperform CLIP's softmax contrastive loss?",
      "difficulty": 3
    },
    {
      "id": 1210,
      "section": "Section 32 — Multimodal AI & Vision-Language Models",
      "question": "How do modern VLMs like Qwen2-VL handle dynamic high-resolution images without excessively expanding the sequence length and destroying the KV cache?",
      "difficulty": 3
    },
    {
      "id": 1211,
      "section": "Section 32 — Multimodal AI & Vision-Language Models",
      "question": "When building a multimodal RAG system for scanned PDF processing, what are the tradeoffs between using a unified VLM versus a two-stage OCR-plus-LLM pipeline?",
      "difficulty": 2
    },
    {
      "id": 1212,
      "section": "Section 32 — Multimodal AI & Vision-Language Models",
      "question": "How does a cross-attention resampler manage varying image token lengths and aspect ratios compared to standard linear projections?",
      "difficulty": 2
    },
    {
      "id": 1213,
      "section": "Section 32 — Multimodal AI & Vision-Language Models",
      "question": "What architectural adaptations are necessary to train a single omni-model (like GPT-4o) that processes native audio alongside text and vision, rather than cascading ASR and LLMs?",
      "difficulty": 3
    },
    {
      "id": 1214,
      "section": "Section 32 — Multimodal AI & Vision-Language Models",
      "question": "How do you mitigate the \"hallucination of object presence\" problem in VLMs when users prompt with leading questions?",
      "difficulty": 2
    },
    {
      "id": 1215,
      "section": "Section 32 — Multimodal AI & Vision-Language Models",
      "question": "Describe the tradeoff between using absolute position embeddings versus 2D-aware RoPE for sub-image tiling in vision encoders.",
      "difficulty": 3
    },
    {
      "id": 1216,
      "section": "Section 32 — Multimodal AI & Vision-Language Models",
      "question": "In multimodal evaluation, why are traditional metrics like BLEU or CIDEr insufficient, and how do frameworks like LLaVA-Bench or MM-Vet address this gap?",
      "difficulty": 2
    },
    {
      "id": 1217,
      "section": "Section 32 — Multimodal AI & Vision-Language Models",
      "question": "How does ImageBind achieve a joint embedding space across six different modalities without requiring paired data for every possible modality combination?",
      "difficulty": 3
    },
    {
      "id": 1218,
      "section": "Section 32 — Multimodal AI & Vision-Language Models",
      "question": "What is the primary bottleneck when scaling native video understanding in VLMs, and how do spatiotemporal pooling strategies mitigate this issue?",
      "difficulty": 2
    },
    {
      "id": 1219,
      "section": "Section 32 — Multimodal AI & Vision-Language Models",
      "question": "When designing a safety filter for a VLM, why is text-only moderation insufficient for detecting multi-modal jailbreaks, such as typography hidden in images?",
      "difficulty": 2
    },
    {
      "id": 1220,
      "section": "Section 32 — Multimodal AI & Vision-Language Models",
      "question": "Contrast the compute requirements of processing a 4K image using a monolithic Vision Transformer versus a dynamic resolution tiling approach.",
      "difficulty": 2
    },
    {
      "id": 1221,
      "section": "Section 32 — Multimodal AI & Vision-Language Models",
      "question": "How do you design a Multimodal RAG system that effectively retrieves across a unified corpus of text documents, charts, and embedded tables?",
      "difficulty": 3
    },
    {
      "id": 1222,
      "section": "Section 32 — Multimodal AI & Vision-Language Models",
      "question": "What role does the \"any-resolution\" strategy play in preserving layout information for complex document understanding tasks in VLMs?",
      "difficulty": 3
    },
    {
      "id": 1223,
      "section": "Section 32 — Multimodal AI & Vision-Language Models",
      "question": "How do models like Gemini 1.5 handle millions of tokens of video context efficiently compared to traditional frame-by-frame VQA pipelines?",
      "difficulty": 3
    },
    {
      "id": 1224,
      "section": "Section 32 — Multimodal AI & Vision-Language Models",
      "question": "In the context of vision-language alignment, what is the impact of freezing the vision encoder versus unfreezing it during the instruction-tuning phase?",
      "difficulty": 3
    },
    {
      "id": 1225,
      "section": "Section 32 — Multimodal AI & Vision-Language Models",
      "question": "What are the security tradeoffs of using open-source VLMs for parsing untrusted user-uploaded images compared to proprietary APIs?",
      "difficulty": 2
    },
    {
      "id": 1226,
      "section": "Section 32 — Multimodal AI & Vision-Language Models",
      "question": "How do omni-models handle the tokenization of continuous audio signals to maintain prosody, emotion, and speaker identity?",
      "difficulty": 3
    },
    {
      "id": 1227,
      "section": "Section 32 — Multimodal AI & Vision-Language Models",
      "question": "Explain the concept of interleaved image-text training data; why is it critical for in-context learning in VLMs compared to standard image-caption pairs?",
      "difficulty": 3
    },
    {
      "id": 1228,
      "section": "Section 32 — Multimodal AI & Vision-Language Models",
      "question": "When implementing a visual agent for web navigation, what is the architectural tradeoff between predicting raw UI coordinates versus grounded HTML elements?",
      "difficulty": 3
    },
    {
      "id": 1229,
      "section": "Section 32 — Multimodal AI & Vision-Language Models",
      "question": "How do you prevent catastrophic forgetting of text-only reasoning capabilities when instruction-tuning a foundational LLM on multimodal tasks?",
      "difficulty": 3
    },
    {
      "id": 1230,
      "section": "Section 32 — Multimodal AI & Vision-Language Models",
      "question": "What is the advantage of using a hierarchical visual encoder (like Swin Transformer) over a standard ViT in tasks requiring dense pixel-level predictions?",
      "difficulty": 2
    },
    {
      "id": 1231,
      "section": "Section 32 — Multimodal AI & Vision-Language Models",
      "question": "How do recent VLMs address the challenge of reading very small text in complex infographics or wide-format charts?",
      "difficulty": 2
    },
    {
      "id": 1232,
      "section": "Section 32 — Multimodal AI & Vision-Language Models",
      "question": "Evaluate the architectural choice of using discrete visual tokens (via VQ-VAE) versus continuous embeddings in multimodal generative models.",
      "difficulty": 3
    },
    {
      "id": 1233,
      "section": "Section 32 — Multimodal AI & Vision-Language Models",
      "question": "How does multimodal contrastive learning handle false negatives in large batch sizes, and what techniques mitigate this?",
      "difficulty": 3
    },
    {
      "id": 1234,
      "section": "Section 32 — Multimodal AI & Vision-Language Models",
      "question": "What are the primary latency implications when switching from a cascaded text-to-speech architecture to a native voice-in/voice-out LLM?",
      "difficulty": 2
    },
    {
      "id": 1235,
      "section": "Section 32 — Multimodal AI & Vision-Language Models",
      "question": "How do you align a VLM to refuse generating instructions for harmful acts depicted solely in an input image without text context?",
      "difficulty": 3
    },
    {
      "id": 1236,
      "section": "Section 32 — Multimodal AI & Vision-Language Models",
      "question": "Compare the utility of semantic embeddings versus structural graph embeddings for retrieving complex financial tables in Multimodal RAG.",
      "difficulty": 2
    },
    {
      "id": 1237,
      "section": "Section 33 — Fine-Tuning, Adaptation & Model Compression",
      "question": "What is the core mechanism of LoRA, and why does it drastically reduce memory usage compared to full fine-tuning?",
      "difficulty": 1
    },
    {
      "id": 1238,
      "section": "Section 33 — Fine-Tuning, Adaptation & Model Compression",
      "question": "When would you explicitly choose full fine-tuning over parameter-efficient methods like LoRA in an enterprise setting?",
      "difficulty": 2
    },
    {
      "id": 1239,
      "section": "Section 33 — Fine-Tuning, Adaptation & Model Compression",
      "question": "Explain the difference between QLoRA and standard LoRA; what specific innovations allow QLoRA to fine-tune a 70B model on a single GPU?",
      "difficulty": 2
    },
    {
      "id": 1240,
      "section": "Section 33 — Fine-Tuning, Adaptation & Model Compression",
      "question": "How does DoRA (Weight-Decomposed Low-Rank Adaptation) improve upon standard LoRA's learning capacity and directionality?",
      "difficulty": 3
    },
    {
      "id": 1241,
      "section": "Section 33 — Fine-Tuning, Adaptation & Model Compression",
      "question": "What is the decision framework for choosing between RAG, prompt engineering, and fine-tuning for a domain-specific QA application?",
      "difficulty": 2
    },
    {
      "id": 1242,
      "section": "Section 33 — Fine-Tuning, Adaptation & Model Compression",
      "question": "Describe the typical three-stage pipeline (CPT, SFT, RLHF) for adapting a base foundation model to a specific proprietary use case.",
      "difficulty": 2
    },
    {
      "id": 1243,
      "section": "Section 33 — Fine-Tuning, Adaptation & Model Compression",
      "question": "Why is loss masking crucial during Supervised Fine-Tuning (SFT), and what happens if you backpropagate on prompt tokens?",
      "difficulty": 1
    },
    {
      "id": 1244,
      "section": "Section 33 — Fine-Tuning, Adaptation & Model Compression",
      "question": "Contrast Post-Training Quantization (PTQ) and Quantization-Aware Training (QAT) in terms of implementation complexity and final model degradation.",
      "difficulty": 2
    },
    {
      "id": 1245,
      "section": "Section 33 — Fine-Tuning, Adaptation & Model Compression",
      "question": "How does AWQ (Activation-aware Weight Quantization) differ from GPTQ, and when should you prefer one over the other for deploying LLMs?",
      "difficulty": 3
    },
    {
      "id": 1246,
      "section": "Section 33 — Fine-Tuning, Adaptation & Model Compression",
      "question": "What is the significance of the GGUF format in the context of local LLM deployment and model quantization on consumer hardware?",
      "difficulty": 1
    },
    {
      "id": 1247,
      "section": "Section 33 — Fine-Tuning, Adaptation & Model Compression",
      "question": "Why has FP8 emerged as a preferred datatype over INT8 for training and inference in modern architectures like Hopper and Blackwell?",
      "difficulty": 3
    },
    {
      "id": 1248,
      "section": "Section 33 — Fine-Tuning, Adaptation & Model Compression",
      "question": "Explain the concept of model merging using TIES; how does it resolve parameter interference compared to simple weight averaging?",
      "difficulty": 3
    },
    {
      "id": 1249,
      "section": "Section 33 — Fine-Tuning, Adaptation & Model Compression",
      "question": "In DARE (Drop and Rescale), how does sparsifying task-specific weights prior to merging improve the final model's generalized performance?",
      "difficulty": 3
    },
    {
      "id": 1250,
      "section": "Section 33 — Fine-Tuning, Adaptation & Model Compression",
      "question": "What is a \"Model Soup,\" and under what specific hyperparameter tuning conditions does it provide a free performance boost without extra inference cost?",
      "difficulty": 2
    },
    {
      "id": 1251,
      "section": "Section 33 — Fine-Tuning, Adaptation & Model Compression",
      "question": "How does SLERP (Spherical Linear Interpolation) maintain weight geometry better than linear interpolation during the model merging process?",
      "difficulty": 3
    },
    {
      "id": 1252,
      "section": "Section 33 — Fine-Tuning, Adaptation & Model Compression",
      "question": "What is the distinction between on-policy and off-policy data in the context of Knowledge Distillation and RLHF?",
      "difficulty": 3
    },
    {
      "id": 1253,
      "section": "Section 33 — Fine-Tuning, Adaptation & Model Compression",
      "question": "How does structured pruning differ from unstructured pruning in its impact on actual inference hardware acceleration and memory bandwidth?",
      "difficulty": 2
    },
    {
      "id": 1254,
      "section": "Section 33 — Fine-Tuning, Adaptation & Model Compression",
      "question": "What strategies effectively mitigate catastrophic forgetting when continually pretraining an LLM on a new, high-resource language?",
      "difficulty": 3
    },
    {
      "id": 1255,
      "section": "Section 33 — Fine-Tuning, Adaptation & Model Compression",
      "question": "How do you evaluate the quality of an SFT dataset, and why is data diversity often more important than sheer volume?",
      "difficulty": 2
    },
    {
      "id": 1256,
      "section": "Section 33 — Fine-Tuning, Adaptation & Model Compression",
      "question": "When distilling a massive teacher model (e.g., GPT-4) into a smaller student (e.g., Llama-3-8B), what is the tradeoff between mimicking logits versus generating synthetic SFT data?",
      "difficulty": 3
    },
    {
      "id": 1257,
      "section": "Section 33 — Fine-Tuning, Adaptation & Model Compression",
      "question": "What is the primary bottleneck in adapter-based PEFT methods during inference, and how do reparameterization techniques solve this?",
      "difficulty": 2
    },
    {
      "id": 1258,
      "section": "Section 33 — Fine-Tuning, Adaptation & Model Compression",
      "question": "Contrast the data requirements, training stability, and mode collapse risks of DPO (Direct Preference Optimization) versus traditional PPO-based RLHF.",
      "difficulty": 3
    },
    {
      "id": 1259,
      "section": "Section 33 — Fine-Tuning, Adaptation & Model Compression",
      "question": "How does grouped-query attention (GQA) interact with KV cache quantization to maximize throughput in memory-constrained serving environments?",
      "difficulty": 3
    },
    {
      "id": 1260,
      "section": "Section 33 — Fine-Tuning, Adaptation & Model Compression",
      "question": "What is the role of a learning rate warmup and decay schedule in preventing loss spikes during the initial phase of full fine-tuning?",
      "difficulty": 1
    },
    {
      "id": 1261,
      "section": "Section 33 — Fine-Tuning, Adaptation & Model Compression",
      "question": "In parameter-efficient fine-tuning, how does targeting all linear layers (Q, K, V, O, and MLP) with LoRA compare to targeting just attention weights?",
      "difficulty": 2
    },
    {
      "id": 1262,
      "section": "Section 33 — Fine-Tuning, Adaptation & Model Compression",
      "question": "How does extreme quantization (e.g., 2-bit or 1.58-bit ternary models like BitNet) fundamentally alter the matrix multiplication operation during inference?",
      "difficulty": 3
    },
    {
      "id": 1263,
      "section": "Section 33 — Fine-Tuning, Adaptation & Model Compression",
      "question": "What are the risks of mode collapse during RLHF, and how can KL divergence penalties against a reference model help maintain model diversity?",
      "difficulty": 2
    },
    {
      "id": 1264,
      "section": "Section 33 — Fine-Tuning, Adaptation & Model Compression",
      "question": "Why is data deduplication critical before Continued Pretraining (CPT), and how does it affect the model's generalization capabilities?",
      "difficulty": 2
    },
    {
      "id": 1265,
      "section": "Section 33 — Fine-Tuning, Adaptation & Model Compression",
      "question": "How do you intuitively determine the optimal rank (r) and alpha parameter when setting up a LoRA fine-tuning run for a novel task?",
      "difficulty": 2
    },
    {
      "id": 1266,
      "section": "Section 33 — Fine-Tuning, Adaptation & Model Compression",
      "question": "Compare the robustness of INT4 weight-only quantization versus INT8 weight-and-activation quantization for complex mathematical reasoning tasks.",
      "difficulty": 3
    },
    {
      "id": 1267,
      "section": "Section 34 — Responsible AI: Fairness, Bias & Explainability",
      "question": "You are deploying a credit scoring model across different demographic groups. How do you choose between demographic parity and equalized odds, given the impossibility theorem of fairness?",
      "difficulty": 3
    },
    {
      "id": 1268,
      "section": "Section 34 — Responsible AI: Fairness, Bias & Explainability",
      "question": "When applying post-processing fairness interventions like Platt scaling by group, what are the primary risks to model performance and compliance?",
      "difficulty": 2
    },
    {
      "id": 1269,
      "section": "Section 34 — Responsible AI: Fairness, Bias & Explainability",
      "question": "How do you distinguish between representation bias and historical bias in an LLM pre-training dataset, and how do their mitigation strategies differ?",
      "difficulty": 2
    },
    {
      "id": 1270,
      "section": "Section 34 — Responsible AI: Fairness, Bias & Explainability",
      "question": "Explain the core limitation of using SHAP values to explain a model with highly correlated features.",
      "difficulty": 3
    },
    {
      "id": 1271,
      "section": "Section 34 — Responsible AI: Fairness, Bias & Explainability",
      "question": "For a real-time fraud detection system, why might LIME be preferred over global SHAP calculations, and what stability trade-offs does it introduce?",
      "difficulty": 2
    },
    {
      "id": 1272,
      "section": "Section 34 — Responsible AI: Fairness, Bias & Explainability",
      "question": "You find that your LLM performs poorly on stereotype benchmarks (e.g., BBQ, CrowS-Pairs). What is the tradeoff between supervised fine-tuning (SFT) and RLHF to mitigate this?",
      "difficulty": 3
    },
    {
      "id": 1273,
      "section": "Section 34 — Responsible AI: Fairness, Bias & Explainability",
      "question": "How do you practically implement continuous algorithmic auditing for a recommendation system in production?",
      "difficulty": 3
    },
    {
      "id": 1274,
      "section": "Section 34 — Responsible AI: Fairness, Bias & Explainability",
      "question": "In the context of the EU AI Act, what specific documentation artifacts bridge the gap between technical metrics and compliance requirements for a high-risk AI system?",
      "difficulty": 2
    },
    {
      "id": 1275,
      "section": "Section 34 — Responsible AI: Fairness, Bias & Explainability",
      "question": "Describe a scenario where individual fairness (similar individuals get similar predictions) directly conflicts with group fairness metrics.",
      "difficulty": 3
    },
    {
      "id": 1276,
      "section": "Section 34 — Responsible AI: Fairness, Bias & Explainability",
      "question": "What are the key differences in debiasing strategies when using pre-processing (data re-weighting) versus in-processing (adversarial debiasing) techniques?",
      "difficulty": 2
    },
    {
      "id": 1277,
      "section": "Section 34 — Responsible AI: Fairness, Bias & Explainability",
      "question": "How does attention visualization in Transformer models fall short as a reliable tool for causal explainability?",
      "difficulty": 3
    },
    {
      "id": 1278,
      "section": "Section 34 — Responsible AI: Fairness, Bias & Explainability",
      "question": "When generating a Datasheet for Datasets, what metadata is most critical to prevent downstream occupational bias in LLM applications?",
      "difficulty": 2
    },
    {
      "id": 1279,
      "section": "Section 34 — Responsible AI: Fairness, Bias & Explainability",
      "question": "How do you evaluate and mitigate representation bias in a multi-modal vision-language model used for image captioning?",
      "difficulty": 3
    },
    {
      "id": 1280,
      "section": "Section 34 — Responsible AI: Fairness, Bias & Explainability",
      "question": "Explain how calibration across groups interacts with predictive parity, and why both cannot be optimized simultaneously if base rates differ.",
      "difficulty": 3
    },
    {
      "id": 1281,
      "section": "Section 34 — Responsible AI: Fairness, Bias & Explainability",
      "question": "What is the impact of using synthetic data generated by LLMs to balance minority classes in training sets regarding amplified biases?",
      "difficulty": 2
    },
    {
      "id": 1282,
      "section": "Section 34 — Responsible AI: Fairness, Bias & Explainability",
      "question": "How do you design a fairness testing pipeline in CI/CD that handles shifting data distributions without causing excessive false positive alerts?",
      "difficulty": 3
    },
    {
      "id": 1283,
      "section": "Section 34 — Responsible AI: Fairness, Bias & Explainability",
      "question": "In banking models regulated by SR 11-7, how do you balance the need for model explainability with the deployment of highly non-linear architectures like gradient boosted trees?",
      "difficulty": 2
    },
    {
      "id": 1284,
      "section": "Section 34 — Responsible AI: Fairness, Bias & Explainability",
      "question": "Compare the effectiveness of counterfactual fairness testing versus purely observational fairness metrics in a healthcare diagnostic model.",
      "difficulty": 3
    },
    {
      "id": 1285,
      "section": "Section 34 — Responsible AI: Fairness, Bias & Explainability",
      "question": "What are the practical limitations of using concept bottleneck models (CBMs) for explainability in deep learning?",
      "difficulty": 2
    },
    {
      "id": 1286,
      "section": "Section 34 — Responsible AI: Fairness, Bias & Explainability",
      "question": "How can prompt engineering techniques (like system prompts) inadvertently introduce new biases while attempting to mitigate existing ones in LLMs?",
      "difficulty": 3
    },
    {
      "id": 1287,
      "section": "Section 34 — Responsible AI: Fairness, Bias & Explainability",
      "question": "How do you measure and address the \"bias of the evaluator\" when using LLM-as-a-judge for fairness benchmarking?",
      "difficulty": 3
    },
    {
      "id": 1288,
      "section": "Section 34 — Responsible AI: Fairness, Bias & Explainability",
      "question": "What role do model cards play in a federated learning setup where the central server never sees the raw training data?",
      "difficulty": 2
    },
    {
      "id": 1289,
      "section": "Section 34 — Responsible AI: Fairness, Bias & Explainability",
      "question": "How do you handle the ethical dilemma of collecting sensitive demographic attributes strictly for the purpose of bias detection and mitigation?",
      "difficulty": 3
    },
    {
      "id": 1290,
      "section": "Section 34 — Responsible AI: Fairness, Bias & Explainability",
      "question": "Explain the difference in explainability requirements for a B2B SaaS analytics tool versus an automated resume screening system.",
      "difficulty": 2
    },
    {
      "id": 1291,
      "section": "Section 34 — Responsible AI: Fairness, Bias & Explainability",
      "question": "How do you utilize activation patching to localize and explain biased internal representations within a large language model?",
      "difficulty": 3
    },
    {
      "id": 1292,
      "section": "Section 35 — Generative AI: Image, Video & Code Generation",
      "question": "In a diffusion model architecture, how does classifier-free guidance (CFG) balance sample diversity against prompt adherence?",
      "difficulty": 3
    },
    {
      "id": 1293,
      "section": "Section 35 — Generative AI: Image, Video & Code Generation",
      "question": "Compare the inference latency and quality trade-offs between DDPM and latent diffusion models like Stable Diffusion.",
      "difficulty": 2
    },
    {
      "id": 1294,
      "section": "Section 35 — Generative AI: Image, Video & Code Generation",
      "question": "How do Diffusion Transformers (DiT) as seen in Sora maintain temporal coherence across long video generations compared to 3D U-Net architectures?",
      "difficulty": 3
    },
    {
      "id": 1295,
      "section": "Section 35 — Generative AI: Image, Video & Code Generation",
      "question": "When building a code generation agent, what are the architectural advantages of a SWE-Agent approach over a simple ReAct loop for resolving SWE-Bench issues?",
      "difficulty": 3
    },
    {
      "id": 1296,
      "section": "Section 35 — Generative AI: Image, Video & Code Generation",
      "question": "Explain the mechanism of ControlNet in guiding diffusion models, and why it is more robust than simple prompt conditioning for edge maps.",
      "difficulty": 2
    },
    {
      "id": 1297,
      "section": "Section 35 — Generative AI: Image, Video & Code Generation",
      "question": "What are the primary bottlenecks in scaling autoregressive models for high-resolution image generation compared to diffusion models?",
      "difficulty": 3
    },
    {
      "id": 1298,
      "section": "Section 35 — Generative AI: Image, Video & Code Generation",
      "question": "How do you evaluate the hallucination rate of a code generation model like Codex when Pass@k on HumanEval does not capture real-world repository context?",
      "difficulty": 3
    },
    {
      "id": 1299,
      "section": "Section 35 — Generative AI: Image, Video & Code Generation",
      "question": "Describe the role of IP-Adapter in enabling zero-shot image prompting for text-to-image diffusion models without fine-tuning.",
      "difficulty": 2
    },
    {
      "id": 1300,
      "section": "Section 35 — Generative AI: Image, Video & Code Generation",
      "question": "In video generation, how does the choice of latent space (e.g., spatial vs. spatio-temporal autoencoders) affect the model's ability to render high-motion scenes?",
      "difficulty": 3
    },
    {
      "id": 1301,
      "section": "Section 35 — Generative AI: Image, Video & Code Generation",
      "question": "How do you implement robust watermarking (e.g., Tree-Ring watermarks) in diffusion models to ensure provenance without degrading image quality?",
      "difficulty": 2
    },
    {
      "id": 1302,
      "section": "Section 35 — Generative AI: Image, Video & Code Generation",
      "question": "Compare the strengths of GANs, Diffusion Models, and Autoregressive models for real-time virtual avatar generation.",
      "difficulty": 3
    },
    {
      "id": 1303,
      "section": "Section 35 — Generative AI: Image, Video & Code Generation",
      "question": "What are the limitations of Frechet Inception Distance (FID) when evaluating diffusion models trained on highly diverse datasets, and what alternatives exist?",
      "difficulty": 2
    },
    {
      "id": 1304,
      "section": "Section 35 — Generative AI: Image, Video & Code Generation",
      "question": "How do neural radiance fields (NeRFs) integrate with diffusion models in modern text-to-3D pipelines (e.g., DreamFusion)?",
      "difficulty": 3
    },
    {
      "id": 1305,
      "section": "Section 35 — Generative AI: Image, Video & Code Generation",
      "question": "When deploying a code copilot, how do you manage the context window to effectively include relevant cross-file dependencies (RAG for code)?",
      "difficulty": 3
    },
    {
      "id": 1306,
      "section": "Section 35 — Generative AI: Image, Video & Code Generation",
      "question": "What is the impact of the C2PA standard on the architecture of enterprise generative AI pipelines for content creation?",
      "difficulty": 2
    },
    {
      "id": 1307,
      "section": "Section 35 — Generative AI: Image, Video & Code Generation",
      "question": "How does inpainting with diffusion models maintain semantic consistency with the unmasked regions of an image?",
      "difficulty": 2
    },
    {
      "id": 1308,
      "section": "Section 35 — Generative AI: Image, Video & Code Generation",
      "question": "In evaluating code generation, why is execution-based evaluation (like HumanEval) strictly superior to BLEU/CodeBLEU, and what are its security risks?",
      "difficulty": 2
    },
    {
      "id": 1309,
      "section": "Section 35 — Generative AI: Image, Video & Code Generation",
      "question": "Explain how Flow Matching provides a more efficient training paradigm for generative models compared to standard DDPMs.",
      "difficulty": 3
    },
    {
      "id": 1310,
      "section": "Section 35 — Generative AI: Image, Video & Code Generation",
      "question": "How do you mitigate the \"mode collapse\" equivalent in diffusion models when heavily fine-tuning on a small, stylized dataset (e.g., via LoRA)?",
      "difficulty": 3
    },
    {
      "id": 1311,
      "section": "Section 35 — Generative AI: Image, Video & Code Generation",
      "question": "What role do score-based generative models play in bridging the gap between continuous-time stochastic differential equations (SDEs) and diffusion models?",
      "difficulty": 3
    },
    {
      "id": 1312,
      "section": "Section 35 — Generative AI: Image, Video & Code Generation",
      "question": "How do you design the reward model in RLHF for a code generation task where structural correctness and algorithmic efficiency are both priorities?",
      "difficulty": 3
    },
    {
      "id": 1313,
      "section": "Section 35 — Generative AI: Image, Video & Code Generation",
      "question": "In Sora-style models, how are varied aspect ratios and resolutions handled during training without aggressive cropping or padding?",
      "difficulty": 3
    },
    {
      "id": 1314,
      "section": "Section 35 — Generative AI: Image, Video & Code Generation",
      "question": "What are the trade-offs of using discrete VQ-VAEs versus continuous latent spaces for video generation tasks?",
      "difficulty": 2
    },
    {
      "id": 1315,
      "section": "Section 35 — Generative AI: Image, Video & Code Generation",
      "question": "How do you leverage attention-map manipulation (e.g., Prompt-to-Prompt) to achieve zero-shot image editing without retraining the diffusion model?",
      "difficulty": 2
    },
    {
      "id": 1316,
      "section": "Section 35 — Generative AI: Image, Video & Code Generation",
      "question": "Describe the architectural requirements for an autonomous Devin-style agent to securely execute and verify code in a sandboxed environment.",
      "difficulty": 3
    },
    {
      "id": 1317,
      "section": "Section 36 — Speech, Audio & Conversational AI",
      "question": "How does the Whisper architecture differ from traditional HMM-based ASR systems, and why is it so robust to accents?",
      "difficulty": 1
    },
    {
      "id": 1318,
      "section": "Section 36 — Speech, Audio & Conversational AI",
      "question": "What are the key tradeoffs in inference latency and accuracy between CTC (Connectionist Temporal Classification) and attention-based encoder-decoder decoding for ASR?",
      "difficulty": 2
    },
    {
      "id": 1319,
      "section": "Section 36 — Speech, Audio & Conversational AI",
      "question": "Why has the Conformer architecture become a standard in modern speech recognition over pure Transformers?",
      "difficulty": 2
    },
    {
      "id": 1320,
      "section": "Section 36 — Speech, Audio & Conversational AI",
      "question": "How do modern non-autoregressive neural TTS models achieve real-time synthesis compared to older autoregressive models like WaveNet?",
      "difficulty": 1
    },
    {
      "id": 1321,
      "section": "Section 36 — Speech, Audio & Conversational AI",
      "question": "What are the core challenges in achieving zero-shot voice cloning with only 3 seconds of reference audio?",
      "difficulty": 2
    },
    {
      "id": 1322,
      "section": "Section 36 — Speech, Audio & Conversational AI",
      "question": "How do discrete audio token models like VALL-E or Bark approach text-to-speech differently than continuous mel-spectrogram generators?",
      "difficulty": 2
    },
    {
      "id": 1323,
      "section": "Section 36 — Speech, Audio & Conversational AI",
      "question": "What are the primary bottlenecks in building a real-time voice agent, and how do you achieve an end-to-end latency under 500ms?",
      "difficulty": 2
    },
    {
      "id": 1324,
      "section": "Section 36 — Speech, Audio & Conversational AI",
      "question": "In an ASR → LLM → TTS pipeline, how do you manage the latency budget when the LLM's TTFT (Time To First Token) exceeds your real-time constraint?",
      "difficulty": 2
    },
    {
      "id": 1325,
      "section": "Section 36 — Speech, Audio & Conversational AI",
      "question": "How do streaming ASR models handle context compared to batch ASR, and what is the impact on Word Error Rate (WER)?",
      "difficulty": 1
    },
    {
      "id": 1326,
      "section": "Section 36 — Speech, Audio & Conversational AI",
      "question": "What techniques are used in speaker diarization to solve the \"who spoke when\" problem in overlapping multi-speaker environments?",
      "difficulty": 2
    },
    {
      "id": 1327,
      "section": "Section 36 — Speech, Audio & Conversational AI",
      "question": "Why is robust Voice Activity Detection (VAD) critical for the performance of a conversational AI agent, and what signals do modern VADs use?",
      "difficulty": 1
    },
    {
      "id": 1328,
      "section": "Section 36 — Speech, Audio & Conversational AI",
      "question": "How does audio classification (e.g., environmental sound detection) differ from ASR in terms of input representations and model architectures?",
      "difficulty": 1
    },
    {
      "id": 1329,
      "section": "Section 36 — Speech, Audio & Conversational AI",
      "question": "What are the unique challenges in modeling long-range temporal dependencies for music understanding and generation?",
      "difficulty": 2
    },
    {
      "id": 1330,
      "section": "Section 36 — Speech, Audio & Conversational AI",
      "question": "In dialogue management, how has the transition from traditional rule-based slot-filling to LLM-driven state tracking impacted system reliability?",
      "difficulty": 2
    },
    {
      "id": 1331,
      "section": "Section 36 — Speech, Audio & Conversational AI",
      "question": "How do you evaluate the accuracy of dialogue state tracking in a complex, multi-turn conversational AI system?",
      "difficulty": 2
    },
    {
      "id": 1332,
      "section": "Section 36 — Speech, Audio & Conversational AI",
      "question": "Why is speech tokenization crucial for building omni-modal foundation models (e.g., GPT-4o)?",
      "difficulty": 2
    },
    {
      "id": 1333,
      "section": "Section 36 — Speech, Audio & Conversational AI",
      "question": "How do neural audio codecs like EnCodec and SoundStream balance bitrate, reconstruction quality, and latency?",
      "difficulty": 2
    },
    {
      "id": 1334,
      "section": "Section 36 — Speech, Audio & Conversational AI",
      "question": "Why is Word Error Rate (WER) sometimes a misleading metric for evaluating ASR in downstream NLP tasks?",
      "difficulty": 1
    },
    {
      "id": 1335,
      "section": "Section 36 — Speech, Audio & Conversational AI",
      "question": "How is the Mean Opinion Score (MOS) collected and used for evaluating TTS, and what are its limitations?",
      "difficulty": 1
    },
    {
      "id": 1336,
      "section": "Section 36 — Speech, Audio & Conversational AI",
      "question": "What metrics are used to quantitatively evaluate speaker similarity in voice cloning, apart from human MOS?",
      "difficulty": 2
    },
    {
      "id": 1337,
      "section": "Section 37 — Edge AI, On-Device ML & Federated Learning",
      "question": "When deciding between TFLite, CoreML, and ExecuTorch for on-device inference, what key hardware and ecosystem factors drive your choice?",
      "difficulty": 2
    },
    {
      "id": 1338,
      "section": "Section 37 — Edge AI, On-Device ML & Federated Learning",
      "question": "How does ONNX Runtime optimize execution graphs differently across heterogeneous edge hardware compared to framework-native runtimes?",
      "difficulty": 2
    },
    {
      "id": 1339,
      "section": "Section 37 — Edge AI, On-Device ML & Federated Learning",
      "question": "What are the accuracy and performance tradeoffs of INT4 vs INT8 quantization for LLM deployment on mobile devices?",
      "difficulty": 3
    },
    {
      "id": 1340,
      "section": "Section 37 — Edge AI, On-Device ML & Federated Learning",
      "question": "How does structured pruning differ from unstructured pruning, and why is structured pruning often preferred for edge deployments?",
      "difficulty": 2
    },
    {
      "id": 1341,
      "section": "Section 37 — Edge AI, On-Device ML & Federated Learning",
      "question": "In what scenarios does Neural Architecture Search (NAS) provide a meaningful ROI for edge AI models compared to manually scaling down architectures?",
      "difficulty": 3
    },
    {
      "id": 1342,
      "section": "Section 37 — Edge AI, On-Device ML & Federated Learning",
      "question": "What is the standard Federated Learning architecture, and how does it guarantee that user data never leaves the device?",
      "difficulty": 2
    },
    {
      "id": 1343,
      "section": "Section 37 — Edge AI, On-Device ML & Federated Learning",
      "question": "What are the key vulnerabilities in Federated Learning, and how can gradient inversion attacks compromise privacy?",
      "difficulty": 3
    },
    {
      "id": 1344,
      "section": "Section 37 — Edge AI, On-Device ML & Federated Learning",
      "question": "How does Differential Privacy mathematically define epsilon and delta, and how do they map to real-world privacy risks?",
      "difficulty": 3
    },
    {
      "id": 1345,
      "section": "Section 37 — Edge AI, On-Device ML & Federated Learning",
      "question": "What is the tradeoff between noise calibration in Differential Privacy and the convergence speed of a federated model?",
      "difficulty": 3
    },
    {
      "id": 1346,
      "section": "Section 37 — Edge AI, On-Device ML & Federated Learning",
      "question": "How do you design an ML pipeline for latency-constrained deployments (e.g., AR glasses) where frame-rate drops induce motion sickness?",
      "difficulty": 3
    },
    {
      "id": 1347,
      "section": "Section 37 — Edge AI, On-Device ML & Federated Learning",
      "question": "What are the primary memory and compute constraints when deploying anomaly detection ML on ultra-low-power IoT microcontrollers?",
      "difficulty": 2
    },
    {
      "id": 1348,
      "section": "Section 37 — Edge AI, On-Device ML & Federated Learning",
      "question": "How do automotive ML constraints differ from mobile ML constraints, particularly regarding safety-critical determinism and thermal throttling?",
      "difficulty": 3
    },
    {
      "id": 1349,
      "section": "Section 37 — Edge AI, On-Device ML & Federated Learning",
      "question": "In an edge-cloud hybrid architecture, what heuristics determine whether to process an inference request locally versus offloading it to the cloud?",
      "difficulty": 2
    },
    {
      "id": 1350,
      "section": "Section 37 — Edge AI, On-Device ML & Federated Learning",
      "question": "How do you handle fallback strategies in hybrid ML architectures when the network connection degrades mid-inference?",
      "difficulty": 2
    },
    {
      "id": 1351,
      "section": "Section 37 — Edge AI, On-Device ML & Federated Learning",
      "question": "Why are Secure Enclaves (TEEs) necessary for certain on-device ML workloads, and what are the performance penalties of using them?",
      "difficulty": 3
    },
    {
      "id": 1352,
      "section": "Section 37 — Edge AI, On-Device ML & Federated Learning",
      "question": "How do NPUs (Neural Processing Units) or Apple's Neural Engine achieve higher efficiency than GPUs for specific tensor operations?",
      "difficulty": 2
    },
    {
      "id": 1353,
      "section": "Section 37 — Edge AI, On-Device ML & Federated Learning",
      "question": "What is a hardware delegate in TFLite/ExecuTorch, and how do you debug a model that falls back to CPU execution instead of using the NPU?",
      "difficulty": 3
    },
    {
      "id": 1354,
      "section": "Section 37 — Edge AI, On-Device ML & Federated Learning",
      "question": "What are the memory and compute challenges of performing on-device fine-tuning (e.g., LoRA) on a mobile phone?",
      "difficulty": 3
    },
    {
      "id": 1355,
      "section": "Section 37 — Edge AI, On-Device ML & Federated Learning",
      "question": "How do you balance personalization and model drift when continuously fine-tuning a model on edge devices?",
      "difficulty": 2
    },
    {
      "id": 1356,
      "section": "Section 37 — Edge AI, On-Device ML & Federated Learning",
      "question": "How do battery life and thermal constraints dictate the batch size and scheduling of on-device ML workloads in mobile OS backgrounds?",
      "difficulty": 3
    },
    {
      "id": 1357,
      "section": "Section 38 — Distributed Training & Large-Scale ML Infrastructure",
      "question": "What is the fundamental difference between Data Parallelism and Model Parallelism, and at what scale does DP become a bottleneck?",
      "difficulty": 2
    },
    {
      "id": 1358,
      "section": "Section 38 — Distributed Training & Large-Scale ML Infrastructure",
      "question": "In Pipeline Parallelism, what causes the \"pipeline bubble,\" and how do micro-batching schedules mitigate it?",
      "difficulty": 3
    },
    {
      "id": 1359,
      "section": "Section 38 — Distributed Training & Large-Scale ML Infrastructure",
      "question": "How does Tensor Parallelism partition a Transformer block, and why is it typically restricted to intra-node GPUs?",
      "difficulty": 3
    },
    {
      "id": 1360,
      "section": "Section 38 — Distributed Training & Large-Scale ML Infrastructure",
      "question": "What is the difference between standard Data Parallelism and Fully Sharded Data Parallel (FSDP)?",
      "difficulty": 2
    },
    {
      "id": 1361,
      "section": "Section 38 — Distributed Training & Large-Scale ML Infrastructure",
      "question": "How do DeepSpeed ZeRO Stages 1 and 2 partition optimizer states and gradients, and what is the communication overhead?",
      "difficulty": 3
    },
    {
      "id": 1362,
      "section": "Section 38 — Distributed Training & Large-Scale ML Infrastructure",
      "question": "Why is DeepSpeed ZeRO Stage 3 considered equivalent to FSDP, and what does it shard that Stage 2 does not?",
      "difficulty": 3
    },
    {
      "id": 1363,
      "section": "Section 38 — Distributed Training & Large-Scale ML Infrastructure",
      "question": "What is the Megatron-LM 3D parallelism pattern, and how does it combine TP, PP, and DP across a massive cluster?",
      "difficulty": 3
    },
    {
      "id": 1364,
      "section": "Section 38 — Distributed Training & Large-Scale ML Infrastructure",
      "question": "Why is Mixed Precision Training (FP16/BF16) essential for large models, and why is BF16 often preferred over FP16 despite having less precision?",
      "difficulty": 2
    },
    {
      "id": 1365,
      "section": "Section 38 — Distributed Training & Large-Scale ML Infrastructure",
      "question": "How does loss scaling prevent underflow in FP16 training, and why is it generally unnecessary when using BF16?",
      "difficulty": 2
    },
    {
      "id": 1366,
      "section": "Section 38 — Distributed Training & Large-Scale ML Infrastructure",
      "question": "What are the unique numerical stability challenges introduced by FP8 training in modern architectures like Hopper?",
      "difficulty": 3
    },
    {
      "id": 1367,
      "section": "Section 38 — Distributed Training & Large-Scale ML Infrastructure",
      "question": "How does gradient accumulation conceptually decouple your mathematical batch size from your physical GPU memory limit?",
      "difficulty": 2
    },
    {
      "id": 1368,
      "section": "Section 38 — Distributed Training & Large-Scale ML Infrastructure",
      "question": "What is the exact compute-versus-memory tradeoff involved in gradient checkpointing (activation recomputation)?",
      "difficulty": 2
    },
    {
      "id": 1369,
      "section": "Section 38 — Distributed Training & Large-Scale ML Infrastructure",
      "question": "How do communication backends like NCCL orchestrate an AllReduce operation across a multi-node GPU cluster?",
      "difficulty": 3
    },
    {
      "id": 1370,
      "section": "Section 38 — Distributed Training & Large-Scale ML Infrastructure",
      "question": "Why is the distinction between AllReduce and AllGather critical when designing communication-efficient parallelism strategies like FSDP?",
      "difficulty": 3
    },
    {
      "id": 1371,
      "section": "Section 38 — Distributed Training & Large-Scale ML Infrastructure",
      "question": "How do modern training infrastructures handle fault tolerance when a single GPU fails in a 10,000 GPU cluster?",
      "difficulty": 3
    },
    {
      "id": 1372,
      "section": "Section 38 — Distributed Training & Large-Scale ML Infrastructure",
      "question": "What are the complexities of elastic training, and how do you dynamically resize a training job without losing state?",
      "difficulty": 3
    },
    {
      "id": 1373,
      "section": "Section 38 — Distributed Training & Large-Scale ML Infrastructure",
      "question": "At extreme scales, how do you prevent the pre-training data pipeline (tokenization, shuffling, loading) from starving the GPUs?",
      "difficulty": 2
    },
    {
      "id": 1374,
      "section": "Section 38 — Distributed Training & Large-Scale ML Infrastructure",
      "question": "How do you calculate the exact GPU memory budget required to train a 70B parameter model using AdamW and mixed precision?",
      "difficulty": 3
    },
    {
      "id": 1375,
      "section": "Section 38 — Distributed Training & Large-Scale ML Infrastructure",
      "question": "What is the Roofline Model, and how does it help identify whether an operation is memory-bandwidth bound or compute-bound?",
      "difficulty": 3
    },
    {
      "id": 1376,
      "section": "Section 38 — Distributed Training & Large-Scale ML Infrastructure",
      "question": "Why is High Bandwidth Memory (HBM) capacity and bandwidth often the true bottleneck in LLM training rather than raw FLOPS?",
      "difficulty": 3
    },
    {
      "id": 1377,
      "section": "Section 39 — Data-Centric AI, Labeling & Synthetic Data",
      "question": "How do you detect and handle label noise when crowd-sourcing annotations for an image classification task?",
      "difficulty": 1
    },
    {
      "id": 1378,
      "section": "Section 39 — Data-Centric AI, Labeling & Synthetic Data",
      "question": "What are the key differences between programmatic weak supervision (e.g., Snorkel) and traditional active learning?",
      "difficulty": 2
    },
    {
      "id": 1379,
      "section": "Section 39 — Data-Centric AI, Labeling & Synthetic Data",
      "question": "Explain the cold start problem in active learning and how to mitigate it in a production setting.",
      "difficulty": 1
    },
    {
      "id": 1380,
      "section": "Section 39 — Data-Centric AI, Labeling & Synthetic Data",
      "question": "How would you use a strong LLM (model-as-judge) to filter synthetic instruction-tuning data?",
      "difficulty": 2
    },
    {
      "id": 1381,
      "section": "Section 39 — Data-Centric AI, Labeling & Synthetic Data",
      "question": "What is Evol-Instruct, and how does it improve the quality of synthetic data for LLM alignment?",
      "difficulty": 2
    },
    {
      "id": 1382,
      "section": "Section 39 — Data-Centric AI, Labeling & Synthetic Data",
      "question": "How do you design a data flywheel for an autonomous driving perception system?",
      "difficulty": 2
    },
    {
      "id": 1383,
      "section": "Section 39 — Data-Centric AI, Labeling & Synthetic Data",
      "question": "What are the trade-offs of using uncertainty sampling versus diversity sampling in active learning?",
      "difficulty": 2
    },
    {
      "id": 1384,
      "section": "Section 39 — Data-Centric AI, Labeling & Synthetic Data",
      "question": "How do you implement curriculum learning to optimize the training trajectory of a large language model?",
      "difficulty": 2
    },
    {
      "id": 1385,
      "section": "Section 39 — Data-Centric AI, Labeling & Synthetic Data",
      "question": "Explain how MinHash LSH is used for large-scale dataset deduplication prior to pre-training.",
      "difficulty": 1
    },
    {
      "id": 1386,
      "section": "Section 39 — Data-Centric AI, Labeling & Synthetic Data",
      "question": "Why is perplexity filtering commonly used in building pre-training corpora for LLMs, and what are its failure modes?",
      "difficulty": 2
    },
    {
      "id": 1387,
      "section": "Section 39 — Data-Centric AI, Labeling & Synthetic Data",
      "question": "How do you ensure benchmark decontamination when creating a new massive pre-training dataset?",
      "difficulty": 2
    },
    {
      "id": 1388,
      "section": "Section 39 — Data-Centric AI, Labeling & Synthetic Data",
      "question": "What strategies would you use to augment tabular data with highly skewed continuous features?",
      "difficulty": 1
    },
    {
      "id": 1389,
      "section": "Section 39 — Data-Centric AI, Labeling & Synthetic Data",
      "question": "How do tools like DVC or lakeFS solve the data versioning problem differently than simply hashing data files?",
      "difficulty": 2
    },
    {
      "id": 1390,
      "section": "Section 39 — Data-Centric AI, Labeling & Synthetic Data",
      "question": "In weak supervision, how do you resolve conflicts when multiple labeling functions disagree on a data point?",
      "difficulty": 2
    },
    {
      "id": 1391,
      "section": "Section 39 — Data-Centric AI, Labeling & Synthetic Data",
      "question": "What is the impact of domain shift on synthetically generated image data, and how do you measure it?",
      "difficulty": 2
    },
    {
      "id": 1392,
      "section": "Section 39 — Data-Centric AI, Labeling & Synthetic Data",
      "question": "How do you evaluate the quality of a synthetically generated dataset before training a production model on it?",
      "difficulty": 2
    },
    {
      "id": 1393,
      "section": "Section 39 — Data-Centric AI, Labeling & Synthetic Data",
      "question": "Explain the concept of query-by-committee in active learning and when you would prefer it.",
      "difficulty": 1
    },
    {
      "id": 1394,
      "section": "Section 39 — Data-Centric AI, Labeling & Synthetic Data",
      "question": "What are the common pitfalls when using Self-Instruct to generate synthetic conversational data?",
      "difficulty": 1
    },
    {
      "id": 1395,
      "section": "Section 39 — Data-Centric AI, Labeling & Synthetic Data",
      "question": "How do you scale data labeling for highly specialized domains like medical imaging where experts are scarce?",
      "difficulty": 2
    },
    {
      "id": 1396,
      "section": "Section 39 — Data-Centric AI, Labeling & Synthetic Data",
      "question": "What is the role of data slicing in discovering and mitigating hidden biases in a training dataset?",
      "difficulty": 2
    },
    {
      "id": 1397,
      "section": "Section 40 — Search, Ranking & Information Retrieval",
      "question": "Compare and contrast pointwise, pairwise, and listwise approaches in Learning to Rank (LTR).",
      "difficulty": 2
    },
    {
      "id": 1398,
      "section": "Section 40 — Search, Ranking & Information Retrieval",
      "question": "How do you design an intent classification system for an e-commerce search engine handling broad queries?",
      "difficulty": 2
    },
    {
      "id": 1399,
      "section": "Section 40 — Search, Ranking & Information Retrieval",
      "question": "Explain the Reciprocal Rank Fusion (RRF) algorithm and why it is effective for hybrid search.",
      "difficulty": 2
    },
    {
      "id": 1400,
      "section": "Section 40 — Search, Ranking & Information Retrieval",
      "question": "In a multi-stage search architecture, how do you balance the trade-off between recall at the retrieval stage and latency at the ranking stage?",
      "difficulty": 3
    },
    {
      "id": 1401,
      "section": "Section 40 — Search, Ranking & Information Retrieval",
      "question": "What is ColBERT's late interaction mechanism, and how does it improve upon traditional cross-encoder re-ranking?",
      "difficulty": 3
    },
    {
      "id": 1402,
      "section": "Section 40 — Search, Ranking & Information Retrieval",
      "question": "How do you optimize BM25 parameters for a domain-specific enterprise search application?",
      "difficulty": 2
    },
    {
      "id": 1403,
      "section": "Section 40 — Search, Ranking & Information Retrieval",
      "question": "How do you evaluate a search system where users frequently abandon searches without clicking any results?",
      "difficulty": 3
    },
    {
      "id": 1404,
      "section": "Section 40 — Search, Ranking & Information Retrieval",
      "question": "Explain the differences between NDCG and MRR, and state when you would prioritize one over the other.",
      "difficulty": 2
    },
    {
      "id": 1405,
      "section": "Section 40 — Search, Ranking & Information Retrieval",
      "question": "How do you handle vocabulary mismatch between queries and documents in semantic search without RAG?",
      "difficulty": 2
    },
    {
      "id": 1406,
      "section": "Section 40 — Search, Ranking & Information Retrieval",
      "question": "Design a real-time personalized ranking system for a news aggregator.",
      "difficulty": 3
    },
    {
      "id": 1407,
      "section": "Section 40 — Search, Ranking & Information Retrieval",
      "question": "How do you implement query expansion using LLMs while maintaining tight latency constraints?",
      "difficulty": 3
    },
    {
      "id": 1408,
      "section": "Section 40 — Search, Ranking & Information Retrieval",
      "question": "What are the challenges of serving a dense vector index at billion-scale, and how do you partition it?",
      "difficulty": 3
    },
    {
      "id": 1409,
      "section": "Section 40 — Search, Ranking & Information Retrieval",
      "question": "How do you detect and handle seasonal or trending queries in an e-commerce search system?",
      "difficulty": 2
    },
    {
      "id": 1410,
      "section": "Section 40 — Search, Ranking & Information Retrieval",
      "question": "Explain how you would implement query rewriting to improve recall for long-tail, conversational queries.",
      "difficulty": 2
    },
    {
      "id": 1411,
      "section": "Section 40 — Search, Ranking & Information Retrieval",
      "question": "How does a cross-encoder differ from a bi-encoder in terms of architecture and production serving costs?",
      "difficulty": 2
    },
    {
      "id": 1412,
      "section": "Section 40 — Search, Ranking & Information Retrieval",
      "question": "How do you account for position bias when training a learning-to-rank model on historical click logs?",
      "difficulty": 3
    },
    {
      "id": 1413,
      "section": "Section 40 — Search, Ranking & Information Retrieval",
      "question": "What is the role of caching in a hybrid search pipeline, and what cache eviction policies work best?",
      "difficulty": 2
    },
    {
      "id": 1414,
      "section": "Section 40 — Search, Ranking & Information Retrieval",
      "question": "How do you design an evaluation framework to measure the impact of search personalization?",
      "difficulty": 3
    },
    {
      "id": 1415,
      "section": "Section 40 — Search, Ranking & Information Retrieval",
      "question": "Explain how to use click models (e.g., Cascade Model, DBN) to extract unbiased relevance labels.",
      "difficulty": 3
    },
    {
      "id": 1416,
      "section": "Section 40 — Search, Ranking & Information Retrieval",
      "question": "How do you handle multi-lingual search in an enterprise setting without maintaining separate indexes per language?",
      "difficulty": 3
    },
    {
      "id": 1417,
      "section": "Section 41 — Causal Inference & Experimentation",
      "question": "How do you use d-separation in a causal DAG to determine which variables to control for in an observational study?",
      "difficulty": 2
    },
    {
      "id": 1418,
      "section": "Section 41 — Causal Inference & Experimentation",
      "question": "What is the difference between ATE, ATT, and ATC, and when would you optimize for ATT over ATE?",
      "difficulty": 3
    },
    {
      "id": 1419,
      "section": "Section 41 — Causal Inference & Experimentation",
      "question": "Explain how Propensity Score Matching (PSM) reduces confounding bias in non-randomized data.",
      "difficulty": 2
    },
    {
      "id": 1420,
      "section": "Section 41 — Causal Inference & Experimentation",
      "question": "When using Inverse Probability Weighting (IPW), how do you handle extreme weights that destabilize variance?",
      "difficulty": 3
    },
    {
      "id": 1421,
      "section": "Section 41 — Causal Inference & Experimentation",
      "question": "How do you use Instrumental Variables (IV) to estimate causal effects in the presence of unobserved confounders?",
      "difficulty": 3
    },
    {
      "id": 1422,
      "section": "Section 41 — Causal Inference & Experimentation",
      "question": "Explain the Regression Discontinuity Design (RDD) and provide an example in a tech industry setting.",
      "difficulty": 2
    },
    {
      "id": 1423,
      "section": "Section 41 — Causal Inference & Experimentation",
      "question": "What are the advantages of using a T-learner versus an S-learner for estimating Heterogeneous Treatment Effects (HTE)?",
      "difficulty": 3
    },
    {
      "id": 1424,
      "section": "Section 41 — Causal Inference & Experimentation",
      "question": "How do causal forests extend random forests to estimate treatment effects at the individual level?",
      "difficulty": 3
    },
    {
      "id": 1425,
      "section": "Section 41 — Causal Inference & Experimentation",
      "question": "Explain the Difference-in-Differences (DiD) method and the importance of the parallel trends assumption.",
      "difficulty": 2
    },
    {
      "id": 1426,
      "section": "Section 41 — Causal Inference & Experimentation",
      "question": "How do you use synthetic controls to evaluate the impact of a market-level feature launch?",
      "difficulty": 3
    },
    {
      "id": 1427,
      "section": "Section 41 — Causal Inference & Experimentation",
      "question": "What is an uplift model, and how does it differ from a standard churn prediction model?",
      "difficulty": 2
    },
    {
      "id": 1428,
      "section": "Section 41 — Causal Inference & Experimentation",
      "question": "Give an example of when causal inference is strictly necessary and correlation-based ML will fail.",
      "difficulty": 2
    },
    {
      "id": 1429,
      "section": "Section 41 — Causal Inference & Experimentation",
      "question": "How do you detect and mitigate network interference (SUTVA violations) in a social network A/B test?",
      "difficulty": 3
    },
    {
      "id": 1430,
      "section": "Section 41 — Causal Inference & Experimentation",
      "question": "Explain switchback testing and when you would use it over standard A/B testing in a ride-sharing marketplace.",
      "difficulty": 3
    },
    {
      "id": 1431,
      "section": "Section 41 — Causal Inference & Experimentation",
      "question": "How do you perform counterfactual evaluation for a recommender system using logged propensity scores?",
      "difficulty": 3
    },
    {
      "id": 1432,
      "section": "Section 42 — Graph ML & Knowledge Graphs",
      "question": "Explain the message-passing framework in Graph Neural Networks (GNNs).",
      "difficulty": 2
    },
    {
      "id": 1433,
      "section": "Section 42 — Graph ML & Knowledge Graphs",
      "question": "How does GraphSAGE address the scalability limitations of standard Graph Convolutional Networks (GCNs)?",
      "difficulty": 3
    },
    {
      "id": 1434,
      "section": "Section 42 — Graph ML & Knowledge Graphs",
      "question": "Compare node2vec with standard matrix factorization techniques for graph representation learning.",
      "difficulty": 2
    },
    {
      "id": 1435,
      "section": "Section 42 — Graph ML & Knowledge Graphs",
      "question": "How does a Graph Attention Network (GAT) compute attention coefficients, and why is this useful for heterogeneous graphs?",
      "difficulty": 3
    },
    {
      "id": 1436,
      "section": "Section 42 — Graph ML & Knowledge Graphs",
      "question": "Explain the architecture of Graph RAG and when it provides more value than standard vector-based RAG.",
      "difficulty": 3
    },
    {
      "id": 1437,
      "section": "Section 42 — Graph ML & Knowledge Graphs",
      "question": "What are the key differences between TransE and RotatE for knowledge graph embedding?",
      "difficulty": 3
    },
    {
      "id": 1438,
      "section": "Section 42 — Graph ML & Knowledge Graphs",
      "question": "How do you implement entity resolution and linking when constructing a knowledge graph from unstructured text?",
      "difficulty": 2
    },
    {
      "id": 1439,
      "section": "Section 42 — Graph ML & Knowledge Graphs",
      "question": "Explain how Cluster-GCN enables large-scale graph training without massive memory overhead.",
      "difficulty": 3
    },
    {
      "id": 1440,
      "section": "Section 42 — Graph ML & Knowledge Graphs",
      "question": "How do you design a fraud detection system using a graph database and real-time transaction data?",
      "difficulty": 3
    },
    {
      "id": 1441,
      "section": "Section 42 — Graph ML & Knowledge Graphs",
      "question": "What is the role of neighbor sampling in training GNNs on massive graphs like social networks?",
      "difficulty": 2
    },
    {
      "id": 1442,
      "section": "Section 42 — Graph ML & Knowledge Graphs",
      "question": "How do you evaluate the quality of a newly constructed enterprise knowledge graph?",
      "difficulty": 2
    },
    {
      "id": 1443,
      "section": "Section 42 — Graph ML & Knowledge Graphs",
      "question": "Explain how to perform link prediction in a bipartite product-user graph for recommendation.",
      "difficulty": 2
    },
    {
      "id": 1444,
      "section": "Section 42 — Graph ML & Knowledge Graphs",
      "question": "When deciding between a property graph (Neo4j) and an RDF graph (Amazon Neptune), what architectural factors do you consider?",
      "difficulty": 3
    },
    {
      "id": 1445,
      "section": "Section 42 — Graph ML & Knowledge Graphs",
      "question": "How do you apply community detection algorithms (e.g., Louvain) to improve content recommendation?",
      "difficulty": 2
    },
    {
      "id": 1446,
      "section": "Section 42 — Graph ML & Knowledge Graphs",
      "question": "What are the failure modes of building a Knowledge Graph automatically using LLM information extraction pipelines?",
      "difficulty": 3
    },
    {
      "id": 1447,
      "section": "Section 43 — Advanced Agentic Systems, Tool Use & Multi-Agent Frameworks",
      "question": "What are the key architectural tradeoffs between ReAct (Reasoning + Acting), Tree-of-Thought (ToT), and Monte Carlo Tree Search (MCTS) for agentic planning, and how do branch evaluation heuristics and compute budget scaling laws govern your choice among them?",
      "difficulty": 2
    },
    {
      "id": 1448,
      "section": "Section 43 — Advanced Agentic Systems, Tool Use & Multi-Agent Frameworks",
      "question": "How does the Toolformer paradigm of self-supervised tool integration differ from downstream zero-shot tool calling via system prompts or JSON Schema, and what are the trade-offs regarding weight baking versus dynamic context overhead?",
      "difficulty": 2
    },
    {
      "id": 1449,
      "section": "Section 43 — Advanced Agentic Systems, Tool Use & Multi-Agent Frameworks",
      "question": "When scaling an agentic system to massive tool registries (1,000+ APIs), how do you architect a retrieval-augmented tool selection (Tool-RAG) and dynamic context injection system without triggering context window exhaustion or tool confusion?",
      "difficulty": 3
    },
    {
      "id": 1450,
      "section": "Section 43 — Advanced Agentic Systems, Tool Use & Multi-Agent Frameworks",
      "question": "How would you design a multi-tiered memory architecture for an autonomous agent combining Working Memory (Context Window), Episodic Memory (Vector/Graph Event Logs), Semantic Memory (Knowledge Graphs/Embeddings), and Procedural Memory (Executable Rules/Tool Protocols)?",
      "difficulty": 2
    },
    {
      "id": 1451,
      "section": "Section 43 — Advanced Agentic Systems, Tool Use & Multi-Agent Frameworks",
      "question": "How do you implement memory consolidation, context compression, and forgetting curves in long-horizon agents to prevent state drift, semantic noise contamination, and exponential token cost scaling?",
      "difficulty": 3
    },
    {
      "id": 1452,
      "section": "Section 43 — Advanced Agentic Systems, Tool Use & Multi-Agent Frameworks",
      "question": "Compare and contrast Graph-Based State Machine orchestration (e.g., LangGraph) against Role-Based / Hierarchical Swarms (e.g., CrewAI / AutoGen). What are the deterministic execution, state transaction isolation, and debugging trade-offs?",
      "difficulty": 2
    },
    {
      "id": 1453,
      "section": "Section 43 — Advanced Agentic Systems, Tool Use & Multi-Agent Frameworks",
      "question": "How do you handle state consensus, deadlocks, and conflicting proposals in a multi-agent swarm where specialized agents disagree on execution plans or code modifications?",
      "difficulty": 3
    },
    {
      "id": 1454,
      "section": "Section 43 — Advanced Agentic Systems, Tool Use & Multi-Agent Frameworks",
      "question": "What is the deep isolation and security architecture required for a Production Code Interpreter agent (e.g., gVisor, Firecracker MicroVMs, eBPF syscall monitoring, cgroups v2, network egress isolation, and AST static analysis)?",
      "difficulty": 3
    },
    {
      "id": 1455,
      "section": "Section 43 — Advanced Agentic Systems, Tool Use & Multi-Agent Frameworks",
      "question": "Explain the mechanics of the Reflexion pattern and iterative critique loops. How do you design a retry state machine with diagnostic memory buffers that prevents infinite self-reflection loops while maximizing task completion rates?",
      "difficulty": 2
    },
    {
      "id": 1456,
      "section": "Section 43 — Advanced Agentic Systems, Tool Use & Multi-Agent Frameworks",
      "question": "How do you construct a Generator-Critic-Refiner pipeline to eliminate hallucinations in tool execution parameters and ensure all agentic claims are anchored to empirical tool responses?",
      "difficulty": 2
    },
    {
      "id": 1457,
      "section": "Section 43 — Advanced Agentic Systems, Tool Use & Multi-Agent Frameworks",
      "question": "How do you benchmark complex agentic workflows using SWE-bench and GAIA, and what metrics beyond Task Success Rate (e.g., trajectory efficiency, pass@k, context drift, cost-per-successful-plan) must you instrument?",
      "difficulty": 3
    },
    {
      "id": 1458,
      "section": "Section 43 — Advanced Agentic Systems, Tool Use & Multi-Agent Frameworks",
      "question": "Design an interruptible Human-in-the-Loop (HITL) escalation state machine that evaluates risk scores, pauses execution, serializes session state, handles asynchronous human approval/edits/rejections, and resumes graph traversal seamlessly.",
      "difficulty": 2
    },
    {
      "id": 1459,
      "section": "Section 43 — Advanced Agentic Systems, Tool Use & Multi-Agent Frameworks",
      "question": "How do you prevent privilege escalation and malicious prompt injection from compromising tool calls when an agent operates with delegate permissions on behalf of an authenticated enterprise user?",
      "difficulty": 2
    },
    {
      "id": 1460,
      "section": "Section 43 — Advanced Agentic Systems, Tool Use & Multi-Agent Frameworks",
      "question": "How do constrained decoding engines (e.g., Pushdown Automata over JSON Grammars like Outlines or XGrammar) enforce 100% structured tool schema adherence at the logit sampling level, and how does this compare to standard Pydantic runtime parsing?",
      "difficulty": 2
    },
    {
      "id": 1461,
      "section": "Section 43 — Advanced Agentic Systems, Tool Use & Multi-Agent Frameworks",
      "question": "How do you architect an auto-repair pipeline to handle schema drift, missing parameters, non-deterministic formatting errors, and unexpected third-party API changes in agentic tool execution?",
      "difficulty": 3
    },
    {
      "id": 1462,
      "section": "Section 43 — Advanced Agentic Systems, Tool Use & Multi-Agent Frameworks",
      "question": "How do you achieve fault-tolerant persistent session state management for multi-turn long-running agents using event sourcing, transactional state checkpointing, and idempotent side-effect execution across infrastructure pod failures?",
      "difficulty": 3
    },
    {
      "id": 1463,
      "section": "Section 43 — Advanced Agentic Systems, Tool Use & Multi-Agent Frameworks",
      "question": "What algorithms and heuristics (e.g., trajectory hashing, semantic similarity thresholding on action reasoning, progress metrics) would you deploy to detect infinite thought loops, state oscillation, and agent stagnation in real time?",
      "difficulty": 2
    },
    {
      "id": 1464,
      "section": "Section 43 — Advanced Agentic Systems, Tool Use & Multi-Agent Frameworks",
      "question": "How does an agent dynamically repair its execution DAG when a tool invocation returns a hard error (e.g., 500 API failure, schema error, or missing file), and what autonomous recovery cascade policies should be enforced before failing?",
      "difficulty": 3
    },
    {
      "id": 1465,
      "section": "Section 43 — Advanced Agentic Systems, Tool Use & Multi-Agent Frameworks",
      "question": "How do you design an asynchronous parallel tool call scheduler that parses model outputs, constructs a dynamic tool dependency DAG, resolves concurrency safety, and executes non-interdependent tools in parallel via an event loop?",
      "difficulty": 2
    },
    {
      "id": 1466,
      "section": "Section 43 — Advanced Agentic Systems, Tool Use & Multi-Agent Frameworks",
      "question": "How do you design a context window token budgeting system that dynamically partitions token headroom between system guardrails, active task state, retrieved episodic memory, and step output scratchpads during deep trajectory execution?",
      "difficulty": 3
    },
    {
      "id": 1467,
      "section": "Section 43 — Advanced Agentic Systems, Tool Use & Multi-Agent Frameworks",
      "question": "What strategy and harness would you build to achieve replayability, trace differential analysis, and regression testing for non-deterministic multi-step agent trajectories?",
      "difficulty": 3
    },
    {
      "id": 1468,
      "section": "Section 43 — Advanced Agentic Systems, Tool Use & Multi-Agent Frameworks",
      "question": "How do you implement OAuth 2.0 token delegation (RFC 8693 Token Exchange) and context-aware access control (ABAC/RBAC) in an agentic framework so tools execute under strict least-privilege scoping?",
      "difficulty": 2
    },
    {
      "id": 1469,
      "section": "Section 43 — Advanced Agentic Systems, Tool Use & Multi-Agent Frameworks",
      "question": "What are the key design principles for an Agent-to-Agent (A2A) protocol covering semantic negotiation, message envelope standards (JSON-RPC / FIPA ACL), performatives (REQUEST, PROPOSE, REJECT), and shared blackboard state stores?",
      "difficulty": 3
    },
    {
      "id": 1470,
      "section": "Section 43 — Advanced Agentic Systems, Tool Use & Multi-Agent Frameworks",
      "question": "How do you implement a tiered model router (combining small fine-tuned models with frontier LLMs) and prompt caching strategies to reduce the token cost of multi-step agent trajectories by 70%+ without dropping task accuracy?",
      "difficulty": 2
    },
    {
      "id": 1471,
      "section": "Section 43 — Advanced Agentic Systems, Tool Use & Multi-Agent Frameworks",
      "question": "Architect a complete, end-to-end enterprise system design for an Autonomous Software Engineering Agent (like Devin / SWE-agent) that ingests GitHub issues, navigates multi-file codebases, executes sandboxed tests, handles human approval, and opens verified pull requests.",
      "difficulty": 3
    },
    {
      "id": 1472,
      "section": "Section 44 — AI Hardware Acceleration, Low-Level Kernels & Compute Engineering",
      "question": "How do physical bandwidth limits and memory hierarchy latency (HBM3e High Bandwidth Memory vs SRAM/L1/L2 caches) create memory bandwidth bottlenecks in LLM inference, and how does memory layout (e.g., contiguity, memory alignment, global memory access coalescing) affect global memory throughput on NVIDIA H100/B200 GPUs? ⭐ Standard",
      "difficulty": 0
    },
    {
      "id": 1473,
      "section": "Section 44 — AI Hardware Acceleration, Low-Level Kernels & Compute Engineering",
      "question": "Explain the CUDA execution model (Grid, Block, Warp, Thread) and its hardware mapping to Streaming Multiprocessors (SMs). How does warp divergence occur at the SIMT execution layer, what is its quantitative penalty on execution latency, and how do CUDA developers mitigate branch divergence using predication and warp-level primitives? ⭐ Standard",
      "difficulty": 0
    },
    {
      "id": 1474,
      "section": "Section 44 — AI Hardware Acceleration, Low-Level Kernels & Compute Engineering",
      "question": "How does OpenAI Triton's programming model abstract CUDA thread/block indexing into block-level parallel programming, and how does Triton's compiler pipeline (JIT, Triton-IR, LLVM-IR, PTX) automatically perform memory coalescing, shared memory allocation, and instruction pipelining compared to hand-written CUDA C++? ⭐ Standard",
      "difficulty": 0
    },
    {
      "id": 1475,
      "section": "Section 44 — AI Hardware Acceleration, Low-Level Kernels & Compute Engineering",
      "question": "Walk through the mathematical formulation of FlashAttention-1. How does tile-based online softmax re-normalize intermediate attention outputs without materializing the full $N \\times N$ attention matrix in HBM, reducing memory complexity from $O(N^2)$ to $O(N)$ and minimizing HBM read/write traffic? ⭐⭐ Hard",
      "difficulty": 0
    },
    {
      "id": 1476,
      "section": "Section 44 — AI Hardware Acceleration, Low-Level Kernels & Compute Engineering",
      "question": "Compare FlashAttention-1, FlashAttention-2, and FlashAttention-3. What key optimizations were introduced in FlashAttention-2 (outer loop over Q vs K/V, non-scaled softmax, sequence parallelism) and FlashAttention-3 (WGMMA / Hopper Tensor Core asynchronous execution, FP8 support, overlapping GEMM with softmax via warp specialization)? ⭐⭐⭐ Principal",
      "difficulty": 0
    },
    {
      "id": 1477,
      "section": "Section 44 — AI Hardware Acceleration, Low-Level Kernels & Compute Engineering",
      "question": "How do NVIDIA Tensor Cores execute low-precision Matrix Multiply-Accumulate (MMA) operations at the hardware instruction level (e.g., `mma.sync` / `wmma`), and what are the quantitative FLOP/s speedups, numerical dynamic range trade-offs, and quantization formats (E4M3 vs E5M2 FP8, INT8 weight-only vs W8A8) when moving from FP16 to FP8/INT8? ⭐⭐ Hard",
      "difficulty": 0
    },
    {
      "id": 1478,
      "section": "Section 44 — AI Hardware Acceleration, Low-Level Kernels & Compute Engineering",
      "question": "Define Arithmetic Intensity (FLOPs per byte transferred) in the context of the Roofline Model. Contrast the prefill (prompt processing) phase and decode (autoregressive generation) phase of LLM inference in terms of arithmetic intensity, HBM bandwidth consumption, and compute-bound vs memory-bound characteristics. ⭐ Standard",
      "difficulty": 0
    },
    {
      "id": 1479,
      "section": "Section 44 — AI Hardware Acceleration, Low-Level Kernels & Compute Engineering",
      "question": "How does Google TPU architecture (v5e/v6 Trillium) differ fundamentally from NVIDIA GPUs regarding Matrix Multiply Units (MXUs), Systolic Arrays, Vector Processing Units (VPUs), and memory architecture? How does a 128x128 systolic array process matrix multiplication with minimal register file transfers? ⭐⭐ Hard",
      "difficulty": 0
    },
    {
      "id": 1480,
      "section": "Section 44 — AI Hardware Acceleration, Low-Level Kernels & Compute Engineering",
      "question": "What are the primary architectural constraints and design principles of mobile/edge NPUs (e.g., Apple Neural Engine, Qualcomm Hexagon, ARM Ethos)? How do fixed SRAM memory budgets, tight power envelopes (TDP < 5W), activation compression, and INT4/INT8 quantization influence on-device model deployment? ⭐ Standard",
      "difficulty": 0
    },
    {
      "id": 1481,
      "section": "Section 44 — AI Hardware Acceleration, Low-Level Kernels & Compute Engineering",
      "question": "Explain the NVLink generation evolution (NVLink-4 on H100 vs NVLink-5 / NVL72 on Blackwell) and NVSwitch topology. How does bi-directional interconnect bandwidth (900 GB/s to 1.8 TB/s per GPU) prevent communication bottlenecks in Tensor Parallelism (All-Reduce) and Pipeline Parallelism compared to PCIe Gen 5/6? ⭐⭐ Hard",
      "difficulty": 0
    },
    {
      "id": 1482,
      "section": "Section 44 — AI Hardware Acceleration, Low-Level Kernels & Compute Engineering",
      "question": "Why is kernel fusion (e.g., RMSNorm + QKV Projection, RoPE + FlashDecoding, SwiGLU fusion) critical for LLM autoregressive decoding? How does custom kernel fusion eliminate intermediate HBM round-trips, reduce CUDA kernel launch overheads, and maximize SRAM reuse during token generation? ⭐⭐ Hard",
      "difficulty": 0
    },
    {
      "id": 1483,
      "section": "Section 44 — AI Hardware Acceleration, Low-Level Kernels & Compute Engineering",
      "question": "Explain CUDA Shared Memory bank conflicts and global memory access coalescing at the warp hardware level. How do 32-bank shared memory architectures handle strided or broadcast memory patterns, and how can padding or memory transpose prevent bank conflicts in GEMM tiling kernels? ⭐⭐ Hard",
      "difficulty": 0
    },
    {
      "id": 1484,
      "section": "Section 44 — AI Hardware Acceleration, Low-Level Kernels & Compute Engineering",
      "question": "How does Tensor Memory Accelerator (TMA) in NVIDIA Hopper/Blackwell hardware abstract and accelerate multi-dimensional tensor copies directly between Global Memory (HBM) and Shared Memory (SRAM) bypassing registers, and how do CUDA Async Barriers and pipeline stages enable double-buffering / compute-copy overlapping? ⭐⭐⭐ Principal",
      "difficulty": 0
    },
    {
      "id": 1485,
      "section": "Section 44 — AI Hardware Acceleration, Low-Level Kernels & Compute Engineering",
      "question": "Draft the step-by-step logic and mathematical decomposition for implementing a custom fused RMSNorm / LayerNorm kernel in Triton. How do block-wise reductions (`tl.reduce`, `tl.sum`) in Triton handle numerical stability (variance estimation, epsilon addition) and vectorization across rows? ⭐⭐ Hard",
      "difficulty": 0
    },
    {
      "id": 1486,
      "section": "Section 44 — AI Hardware Acceleration, Low-Level Kernels & Compute Engineering",
      "question": "Standard FlashAttention optimizes prompt prefill, but suffers low occupancy during single-token autoregressive decoding across long context lengths ($N > 32K$). How does FlashDecoding parallelize the KV sequence dimension across thread blocks, aggregate partial online softmax results, and maintain low latency? ⭐⭐⭐ Principal",
      "difficulty": 0
    },
    {
      "id": 1487,
      "section": "Section 44 — AI Hardware Acceleration, Low-Level Kernels & Compute Engineering",
      "question": "When executing FP8 Tensor Core matrix multiplications (e.g., FP8 E4M3 for weights/activations), how are delayed scaling factors (per-tensor vs per-block scaling) computed and applied? How do low-level CUDA kernels perform fused dequantization and FP16/BF16 output accumulation to avoid dynamic range overflow/underflow? ⭐⭐⭐ Principal",
      "difficulty": 0
    },
    {
      "id": 1488,
      "section": "Section 44 — AI Hardware Acceleration, Low-Level Kernels & Compute Engineering",
      "question": "Apply the Roofline Model to analyze the memory traffic of KV Cache reads during LLM decoding. How does PagedAttention (vLLM) optimize virtual memory management, eliminate external memory fragmentation, and improve effective memory bandwidth utilization near the Roofline ceiling? ⭐⭐ Hard",
      "difficulty": 0
    },
    {
      "id": 1489,
      "section": "Section 44 — AI Hardware Acceleration, Low-Level Kernels & Compute Engineering",
      "question": "How does the XLA (Accelerated Linear Algebra) compiler target Google TPUs by transforming High-Level Optimizer (HLO) computational graphs into executable TPU code? Explain HLO instruction fusion, memory allocation mapping to High Bandwidth Memory vs Vector Memory (VMEM), and loop tiling for systolic arrays. ⭐⭐⭐ Principal",
      "difficulty": 0
    },
    {
      "id": 1490,
      "section": "Section 44 — AI Hardware Acceleration, Low-Level Kernels & Compute Engineering",
      "question": "On edge NPUs with strictly integer execution units (INT8/INT4), how does post-training quantization (PTQ) handle activations with extreme outliers (e.g., SmoothQuant, AWQ)? How does the NPU compiler tile static tensor graphs to fit within tight on-chip SRAM buffers without spilling to LPDDR5 DRAM? ⭐⭐ Hard",
      "difficulty": 0
    },
    {
      "id": 1491,
      "section": "Section 44 — AI Hardware Acceleration, Low-Level Kernels & Compute Engineering",
      "question": "Contrast host-driven NCCL ring/tree collectives with hardware-accelerated NVLS (NVLink Switch System) in-network reduction. How does NVLS perform vector additions directly inside the NVSwitch hardware during All-Reduce, and what is its impact on scaling 70B+ LLM training across 1,000+ GPUs? ⭐⭐⭐ Principal",
      "difficulty": 0
    },
    {
      "id": 1492,
      "section": "Section 44 — AI Hardware Acceleration, Low-Level Kernels & Compute Engineering",
      "question": "Describe the low-level implementation details of a fused Rotary Position Embedding (RoPE) + KV Cache update kernel in Triton/CUDA. Why is fusing complex number rotations with key-value memory writes into a single kernel pass critical for minimizing memory traffic during decoding? ⭐⭐ Hard",
      "difficulty": 0
    },
    {
      "id": 1493,
      "section": "Section 44 — AI Hardware Acceleration, Low-Level Kernels & Compute Engineering",
      "question": "From a hardware acceleration and roofline perspective, analyze Mixture-of-Experts (MoE) layer execution (e.g., Mixtral-8x7B, DeepSeek-V3). Why do sparse MoE layers alter the arithmetic intensity of LLM decoding, and what memory routing bottlenecks arise when transferring expert tokens across GPUs via NVLink? ⭐⭐⭐ Principal",
      "difficulty": 0
    },
    {
      "id": 1494,
      "section": "Section 44 — AI Hardware Acceleration, Low-Level Kernels & Compute Engineering",
      "question": "How do Triton's memory block abstractions (`tl.load`, `tl.store` with indirect block pointers/masks) simplify the authoring of dynamic PagedAttention kernels compared to CUDA indexing logic? Discuss pointer arithmetic, block masking, and atomic reduction for variable sequence lengths. ⭐⭐⭐ Principal",
      "difficulty": 0
    },
    {
      "id": 1495,
      "section": "Section 44 — AI Hardware Acceleration, Low-Level Kernels & Compute Engineering",
      "question": "Compare Weight-Only Quantization (W4A16) and Weight-Activation Quantization (W8A8) from an execution unit standpoint. How do CUDA/Triton kernels dynamically unpack 4-bit weights in register files on Tensor Cores versus how TPU MXUs process INT8 matrix multiplications? ⭐⭐ Hard",
      "difficulty": 0
    },
    {
      "id": 1496,
      "section": "Section 44 — AI Hardware Acceleration, Low-Level Kernels & Compute Engineering",
      "question": "In advanced LLM inference frameworks (e.g., TensorRT-LLM, vLLM, SGLang), how do CUDA Graphs, custom multi-kernel fusion, and speculative decoding verification kernels collaborate to minimize CPU-to-GPU launch overheads and maximize GPU Tensor Core utilization during token-by-token generation? ⭐⭐⭐ Principal",
      "difficulty": 0
    },
    {
      "id": 1497,
      "section": "Section 45 — AI Security, Red Teaming, Adversarial ML & Guardrails",
      "question": "What is the fundamental operational difference between Direct and Indirect Prompt Injection attacks, how does privilege escalation manifest in agentic LLM systems, and what architectural pattern prevents untrusted retrieved data from corrupting execution state?",
      "difficulty": 2
    },
    {
      "id": 1498,
      "section": "Section 45 — AI Security, Red Teaming, Adversarial ML & Guardrails",
      "question": "How do secondary LLM filtering, strict schema validation, control token demarcation, and runtime sandboxing combine to create defense-in-depth against indirect prompt injection in RAG pipelines?",
      "difficulty": 2
    },
    {
      "id": 1499,
      "section": "Section 45 — AI Security, Red Teaming, Adversarial ML & Guardrails",
      "question": "What are the mechanics of Many-Shot Jailbreaking (MSJ) and Refusal Suppression attacks on long-context LLMs, why does safety alignment decay over long context windows, and how do you mitigate them at the platform level?",
      "difficulty": 2
    },
    {
      "id": 1500,
      "section": "Section 45 — AI Security, Red Teaming, Adversarial ML & Guardrails",
      "question": "How do attackers extract system prompts and hijack model personas using encoding tricks or multi-turn roleplay, and what operational controls prevent prompt exfiltration? ⭐ Standard",
      "difficulty": 0
    },
    {
      "id": 1501,
      "section": "Section 45 — AI Security, Red Teaming, Adversarial ML & Guardrails",
      "question": "How do clean-label and dirty-label data poisoning attacks inject dormant backdoors into LLM pretraining or SFT datasets, and how do spectral signatures and influence functions detect them prior to training? ⭐⭐⭐ Principal",
      "difficulty": 0
    },
    {
      "id": 1502,
      "section": "Section 45 — AI Security, Red Teaming, Adversarial ML & Guardrails",
      "question": "What structural properties characterize stealthy backdoor triggers in LLMs, and how do weight auditing methods like Activation Clustering and Neural Cleanse identify backdoored models? ⭐⭐ Hard",
      "difficulty": 0
    },
    {
      "id": 1503,
      "section": "Section 45 — AI Security, Red Teaming, Adversarial ML & Guardrails",
      "question": "How do Likelihood Ratio Attacks (LiRA) and loss distribution analysis execute Membership Inference Attacks (MIA) against foundation models, what are the regulatory implications, and how does Differential Privacy defend against them? ⭐⭐⭐ Principal",
      "difficulty": 0
    },
    {
      "id": 1504,
      "section": "Section 45 — AI Security, Red Teaming, Adversarial ML & Guardrails",
      "question": "How do model inversion attacks extract exact private training inputs from logit confidence distributions or gradient exposures, and how does gradient clipping with noise injection mitigate this risk? ⭐⭐⭐ Principal",
      "difficulty": 0
    },
    {
      "id": 1505,
      "section": "Section 45 — AI Security, Red Teaming, Adversarial ML & Guardrails",
      "question": "What are the architectural components of a deterministic guardrail pipeline (Regex, AST parsing, Pydantic schemas, canary tokens), and how do you achieve sub-millisecond validation latencies? ⭐ Standard",
      "difficulty": 0
    },
    {
      "id": 1506,
      "section": "Section 45 — AI Security, Red Teaming, Adversarial ML & Guardrails",
      "question": "How do dedicated safety models like Llama Guard classify input/output hazards, how do NeMo Guardrails programmatically enforce conversation flows via Colang, and what are the performance latency trade-offs? ⭐⭐ Hard",
      "difficulty": 0
    },
    {
      "id": 1507,
      "section": "Section 45 — AI Security, Red Teaming, Adversarial ML & Guardrails",
      "question": "How do Input, Output, Dialog, and Retrieval Rails operate inside NeMo Guardrails, and how do you prevent circular evaluation loops or tail latency spikes in RAG applications? ⭐⭐ Hard",
      "difficulty": 0
    },
    {
      "id": 1508,
      "section": "Section 45 — AI Security, Red Teaming, Adversarial ML & Guardrails",
      "question": "How do Hardware Enclaves (NVIDIA H100 Confidential Computing, AMD SEV-SNP, Intel TDX) isolate model weights and prompt contexts from compromised host OS/hypervisors, and what is the throughput overhead? ⭐⭐⭐ Principal",
      "difficulty": 0
    },
    {
      "id": 1509,
      "section": "Section 45 — AI Security, Red Teaming, Adversarial ML & Guardrails",
      "question": "Walk through the cryptographic sequence of Remote Attestation, Evidence Verification, and KMS Key Release required to load encrypted model weights into a GPU TEE. ⭐⭐⭐ Principal",
      "difficulty": 0
    },
    {
      "id": 1510,
      "section": "Section 45 — AI Security, Red Teaming, Adversarial ML & Guardrails",
      "question": "How does the Kirchenbauer et al. green-list/red-list token watermarking algorithm work, how do you measure detection significance (z-score), and what is its impact on generation perplexity? ⭐⭐ Hard",
      "difficulty": 0
    },
    {
      "id": 1511,
      "section": "Section 45 — AI Security, Red Teaming, Adversarial ML & Guardrails",
      "question": "How does the C2PA (Coalition for Content Provenance and Authenticity) standard cryptographically bind manifests to AI-generated media, and how do soft-binding vs. hard-binding mechanisms handle metadata stripping? ⭐⭐ Hard",
      "difficulty": 0
    },
    {
      "id": 1512,
      "section": "Section 45 — AI Security, Red Teaming, Adversarial ML & Guardrails",
      "question": "How do white-box (PGD) and black-box adversarial attacks create imperceptible image perturbations that execute visual prompt injections or jailbreak Vision-Language Models (VLMs)? ⭐⭐ Hard",
      "difficulty": 0
    },
    {
      "id": 1513,
      "section": "Section 45 — AI Security, Red Teaming, Adversarial ML & Guardrails",
      "question": "What combination of adversarial training (PGD-AT), feature-space smoothing, and input preprocessing mitigates visual prompt injections without destroying clean image classification accuracy? ⭐⭐ Hard",
      "difficulty": 0
    },
    {
      "id": 1514,
      "section": "Section 45 — AI Security, Red Teaming, Adversarial ML & Guardrails",
      "question": "How do you architect a multi-dimensional (RPM, TPM, CPM) token bucket rate-limiter in Redis using Lua scripts to prevent DoS and context inflation attacks on AI microservices? ⭐ Standard",
      "difficulty": 0
    },
    {
      "id": 1515,
      "section": "Section 45 — AI Security, Red Teaming, Adversarial ML & Guardrails",
      "question": "How do competitors extract proprietary model capabilities via API distillation, and what operational defenses (watermarked outputs, synthetic distribution shifting, logit perturbations) prevent unauthorized model cloning? ⭐⭐ Hard",
      "difficulty": 0
    },
    {
      "id": 1516,
      "section": "Section 45 — AI Security, Red Teaming, Adversarial ML & Guardrails",
      "question": "How do automated red teaming algorithms like GCG (Greedy Coordinate Gradient) and TAP (Tree-of-Attacks with Pruning) discover safety vulnerabilities, and how do you integrate ART into CI/CD deployment gates? ⭐⭐⭐ Principal",
      "difficulty": 0
    },
    {
      "id": 1517,
      "section": "Section 45 — AI Security, Red Teaming, Adversarial ML & Guardrails",
      "question": "What architectural controls enforce the Principle of Least Privilege for autonomous agents executing dynamic tool calls (Python, SQL, Shell), and how do gVisor microVM sandboxes mitigate remote code execution (RCE)? ⭐⭐⭐ Principal",
      "difficulty": 0
    },
    {
      "id": 1518,
      "section": "Section 45 — AI Security, Red Teaming, Adversarial ML & Guardrails",
      "question": "How do attackers invert high-dimensional vector embeddings to reconstruct original document text, and how do you enforce Document Access Control Lists (ACLs) and vector store sanitization to prevent cross-tenant leakage? ⭐⭐ Hard",
      "difficulty": 0
    },
    {
      "id": 1519,
      "section": "Section 45 — AI Security, Red Teaming, Adversarial ML & Guardrails",
      "question": "How do frequency-domain (DWT-DCT-SVD) audio watermarking and latent-space diffusion watermarking (Tree-Ring, Stable Signature) embed undetectable, tamper-resistant provenance signals into media streams? ⭐⭐ Hard",
      "difficulty": 0
    },
    {
      "id": 1520,
      "section": "Section 45 — AI Security, Red Teaming, Adversarial ML & Guardrails",
      "question": "Formulate the DP-SGD algorithm, define the privacy budget $(\\epsilon, \\delta)$ accounting via Rényi Differential Privacy (RDP), and analyze the memory and utility trade-offs when applying DP-LoRA to LLMs. ⭐⭐⭐ Principal",
      "difficulty": 0
    },
    {
      "id": 1521,
      "section": "Section 45 — AI Security, Red Teaming, Adversarial ML & Guardrails",
      "question": "How do you design a comprehensive Zero-Trust Architecture for an enterprise AI platform spanning data ingestion, fine-tuning, registry signing, runtime TEE inference, cascading guardrails, and compliance audit logging? ⭐⭐⭐ Principal",
      "difficulty": 0
    },
    {
      "id": 1522,
      "section": "Section 46 — Long-Context Mechanics, State Space Models (SSMs) & KV-Cache Optimizations",
      "question": "Compare the computational and memory complexity of standard Transformer self-attention $O(N^2)$ with State Space Models (SSMs like S4 and Mamba) during prefill and generation phases. How do SSMs achieve linear time complexity $O(N)$ and $O(1)$ inference memory?",
      "difficulty": 2
    },
    {
      "id": 1523,
      "section": "Section 46 — Long-Context Mechanics, State Space Models (SSMs) & KV-Cache Optimizations",
      "question": "Explain the mathematical transition from Continuous-Time State Space Models (ODEs) to Discrete-Time SSMs via bilinear (Tustin) discretization, and how HiPPO (High-order Polynomial Projection Operators) memory matrices enable S4 to capture long-range dependencies without vanishing gradients.",
      "difficulty": 3
    },
    {
      "id": 1524,
      "section": "Section 46 — Long-Context Mechanics, State Space Models (SSMs) & KV-Cache Optimizations",
      "question": "What is the core innovation of Mamba (Selective State Space Models) over S4? Explain how Mamba's input-dependent parameterization ($B(x), C(x), \\Delta(x)$) breaks time-invariance, rendering LTI convolutions unusable, and how hardware-aware selective scan algorithm (SRAM vs HBM memory hierarchy) enables fast training.",
      "difficulty": 3
    },
    {
      "id": 1525,
      "section": "Section 46 — Long-Context Mechanics, State Space Models (SSMs) & KV-Cache Optimizations",
      "question": "Explain the architecture of RWKV (Receptance Weighted Key Value) and how it unifies the parallelizable training of Transformers with the constant memory $O(1)$ step-wise inference of RNNs via its spatial/time mixing mechanics.",
      "difficulty": 2
    },
    {
      "id": 1526,
      "section": "Section 46 — Long-Context Mechanics, State Space Models (SSMs) & KV-Cache Optimizations",
      "question": "Explain the dynamic properties of Rotary Position Embeddings (RoPE) and why standard RoPE fails when extrapolating sequence lengths beyond the pre-training context window length $L_{train}$.",
      "difficulty": 2
    },
    {
      "id": 1527,
      "section": "Section 46 — Long-Context Mechanics, State Space Models (SSMs) & KV-Cache Optimizations",
      "question": "Compare Position Interpolation (PI), NTK-aware RoPE scaling, and YaRN (Yet Another RoPE Extension). Explain mathematically how YaRN applies temperature scaling to high-frequency dimensions and interpolation to low-frequency dimensions to prevent attention entropy collapse.",
      "difficulty": 3
    },
    {
      "id": 1528,
      "section": "Section 46 — Long-Context Mechanics, State Space Models (SSMs) & KV-Cache Optimizations",
      "question": "How do linear attention variants (e.g., Fast Weight Programmers, Performer, Linear Transformer) approximate softmax attention $\\text{Softmax}(QK^T)V$ using kernel feature maps $\\phi(Q)\\phi(K)^T V$, and what are their empirical stability and expressivity trade-offs compared to full attention?",
      "difficulty": 2
    },
    {
      "id": 1529,
      "section": "Section 46 — Long-Context Mechanics, State Space Models (SSMs) & KV-Cache Optimizations",
      "question": "Explain how standard LLM serving engines manage KV-cache memory using contiguous allocation (leading to internal and external memory fragmentation), and how PagedAttention (vLLM) resolves this via virtual memory paging.",
      "difficulty": 1
    },
    {
      "id": 1530,
      "section": "Section 46 — Long-Context Mechanics, State Space Models (SSMs) & KV-Cache Optimizations",
      "question": "Walk through the detailed memory layout, block allocation, physical block table mapping, and copy-on-write (CoW) mechanics of PagedAttention during multi-request batching and parallel sampling (e.g., beam search).",
      "difficulty": 2
    },
    {
      "id": 1531,
      "section": "Section 46 — Long-Context Mechanics, State Space Models (SSMs) & KV-Cache Optimizations",
      "question": "Explain RadixAttention as implemented in SGLang. How does a Radix Tree maintain prefix KV-cache blocks across multiple requests, enable automatic prefix sharing, and manage cache eviction policies (e.g., LRU on tree nodes)?",
      "difficulty": 3
    },
    {
      "id": 1532,
      "section": "Section 46 — Long-Context Mechanics, State Space Models (SSMs) & KV-Cache Optimizations",
      "question": "Compare RadixAttention (SGLang) with static prefix caching (vLLM/TGI). What are the structural edge cases (e.g., cache fragmentation, lock contention, token-level matching overhead) when operating dynamic prefix caching under high concurrency?",
      "difficulty": 2
    },
    {
      "id": 1533,
      "section": "Section 46 — Long-Context Mechanics, State Space Models (SSMs) & KV-Cache Optimizations",
      "question": "What is Chunked Prefill (e.g., Sarathi-Serve, vLLM chunked prefill), and how does it solve the problem of request starvation and tail-latency (TPOT) spikes caused by large prefill requests in compute-bound vs memory-bound phases?",
      "difficulty": 2
    },
    {
      "id": 1534,
      "section": "Section 46 — Long-Context Mechanics, State Space Models (SSMs) & KV-Cache Optimizations",
      "question": "Architect a Disaggregated Prefill-Decode Serving Cluster (e.g., Splitwise, Mooncake, DistServe). Explain how separating prefill nodes (compute-bound) and decode nodes (memory-bound) optimizes GPU utilization, and analyze the network bandwidth bottlenecks of transmitting high-volume KV-caches across nodes over PCIe/NVLink/RDMA.",
      "difficulty": 3
    },
    {
      "id": 1535,
      "section": "Section 46 — Long-Context Mechanics, State Space Models (SSMs) & KV-Cache Optimizations",
      "question": "Calculate the exact memory footprint of KV-cache for a model with $L$ layers, $H$ key-value heads, hidden dimension $D_{head}$, context length $N$, batch size $B$, in FP16 precision. How does Grouped-Query Attention (GQA) reduce this footprint relative to MHA?",
      "difficulty": 1
    },
    {
      "id": 1536,
      "section": "Section 46 — Long-Context Mechanics, State Space Models (SSMs) & KV-Cache Optimizations",
      "question": "Explain KV-cache quantization strategies for INT8, FP8 (E4M3 vs E5M2), and INT4 formats. What are key challenges such as asymmetric dynamic range, per-channel vs per-token scale factors, and outlier channels in Keys vs Values?",
      "difficulty": 2
    },
    {
      "id": 1537,
      "section": "Section 46 — Long-Context Mechanics, State Space Models (SSMs) & KV-Cache Optimizations",
      "question": "Compare KIVI (2-bit per-channel Key / per-token Value quantization) with QA-KV and SmoothQuant-KV. How do per-channel quantization schemes handle the high dynamic range of key activations across long context windows?",
      "difficulty": 3
    },
    {
      "id": 1538,
      "section": "Section 46 — Long-Context Mechanics, State Space Models (SSMs) & KV-Cache Optimizations",
      "question": "Explain the Heavy Hitter Oracle (H2O) KV-cache eviction algorithm. How does it maintain cumulative attention scores to retain \"Heavy Hitter\" (H2) tokens and recent local tokens, and why does simple magnitude-based pruning fail?",
      "difficulty": 2
    },
    {
      "id": 1539,
      "section": "Section 46 — Long-Context Mechanics, State Space Models (SSMs) & KV-Cache Optimizations",
      "question": "Explain StreamingLLM and the concept of \"Attention Sinks\". Why do initial prompt tokens absorb a disproportionately large amount of attention score even if they lack semantic importance, and how does keeping initial tokens + a sliding window enable infinite-length streaming generation without retraining?",
      "difficulty": 2
    },
    {
      "id": 1540,
      "section": "Section 46 — Long-Context Mechanics, State Space Models (SSMs) & KV-Cache Optimizations",
      "question": "How does ScaNN / vector quantization apply to KV-cache retrieval compression (e.g., FastGen, SparQ Attention)? Compare static token eviction (H2O) with dynamic, query-dependent KV-cache retrieval during decode iterations.",
      "difficulty": 3
    },
    {
      "id": 1541,
      "section": "Section 46 — Long-Context Mechanics, State Space Models (SSMs) & KV-Cache Optimizations",
      "question": "Explain RingAttention for ultra-long context distributed training and inference. How does it overlap block-wise self-attention computation with peer-to-peer ring communication of Key/Value blocks across GPUs, avoiding the $O(N^2)$ memory bottleneck?",
      "difficulty": 3
    },
    {
      "id": 1542,
      "section": "Section 46 — Long-Context Mechanics, State Space Models (SSMs) & KV-Cache Optimizations",
      "question": "Compare RingAttention with DeepSpeed Ulysses (Sequence Parallelism) and Megatron Context Parallelism (CP). Under what sequence lengths and cluster network topologies (InfiniBand vs RoCE) does RingAttention outperform Ulysses/Megatron CP?",
      "difficulty": 2
    },
    {
      "id": 1543,
      "section": "Section 46 — Long-Context Mechanics, State Space Models (SSMs) & KV-Cache Optimizations",
      "question": "Explain the \"Lost in the Middle\" phenomenon (Liu et al.). Why do LLMs demonstrate high retrieval performance for information located at the beginning or end of a long context window, but suffer severe performance degradation when key information is located in the middle?",
      "difficulty": 1
    },
    {
      "id": 1544,
      "section": "Section 46 — Long-Context Mechanics, State Space Models (SSMs) & KV-Cache Optimizations",
      "question": "What underlying mechanisms contribute to context retrieval position bias (e.g., RoPE positional decay, causal masking asymmetry, softmax concentration on early tokens)? How do models like Gemini 1.5 Pro and Claude 3.5 Sonnet overcome this for 1M+ context lengths?",
      "difficulty": 2
    },
    {
      "id": 1545,
      "section": "Section 46 — Long-Context Mechanics, State Space Models (SSMs) & KV-Cache Optimizations",
      "question": "Design an end-to-end evaluation benchmark for ultra-long context LLMs beyond simple \"Needle in a Haystack\" (NIAH). What are the limitations of single-needle retrieval, and how do multi-needle synthesis, long-dependency reasoning, and state-tracking benchmarks stress-test long-context architectures?",
      "difficulty": 3
    },
    {
      "id": 1546,
      "section": "Section 46 — Long-Context Mechanics, State Space Models (SSMs) & KV-Cache Optimizations",
      "question": "Architect an enterprise production inference system for a 1M-token context application (e.g., analyzing a codebase or regulatory repository). Synthesize choices across model architecture (Hybrid SSM-Transformer vs standard LLM + YaRN), KV-cache management (PagedAttention + RadixAttention + FP8 quantization + Chunked Prefill), disaggregated serving infrastructure, and retrieval fallback strategy.",
      "difficulty": 3
    },
    {
      "id": 1547,
      "section": "Section 47 — Domain-Specific AI Architecture (Robotics, Bio, Finance & Software Agents)",
      "question": "Vision-Language-Action (VLA) Model Architecture & High-Frequency Control Loops",
      "difficulty": 3
    },
    {
      "id": 1548,
      "section": "Section 47 — Domain-Specific AI Architecture (Robotics, Bio, Finance & Software Agents)",
      "question": "Closed-Loop Stability, Distribution Shift, and Safety Guardrails in Embodied AI",
      "difficulty": 2
    },
    {
      "id": 1549,
      "section": "Section 47 — Domain-Specific AI Architecture (Robotics, Bio, Finance & Software Agents)",
      "question": "Cross-Embodiment Generalization & Action Space Unification (Open X-Embodiment)",
      "difficulty": 3
    },
    {
      "id": 1550,
      "section": "Section 47 — Domain-Specific AI Architecture (Robotics, Bio, Finance & Software Agents)",
      "question": "AlphaFold3 Architecture: Pairformer & Unified Diffusion for Biomolecules",
      "difficulty": 3
    },
    {
      "id": 1551,
      "section": "Section 47 — Domain-Specific AI Architecture (Robotics, Bio, Finance & Software Agents)",
      "question": "Equivariant vs. Invariant 3D Diffusion Representations in AlphaFold3",
      "difficulty": 3
    },
    {
      "id": 1552,
      "section": "Section 47 — Domain-Specific AI Architecture (Robotics, Bio, Finance & Software Agents)",
      "question": "Structural Confidence Validation & Disordered Region Handling in AlphaFold3",
      "difficulty": 2
    },
    {
      "id": 1553,
      "section": "Section 47 — Domain-Specific AI Architecture (Robotics, Bio, Finance & Software Agents)",
      "question": "Modeling Protein-Ligand and Protein-RNA Physical Binding Interfaces",
      "difficulty": 2
    },
    {
      "id": 1554,
      "section": "Section 47 — Domain-Specific AI Architecture (Robotics, Bio, Finance & Software Agents)",
      "question": "Clinical LLMs & Medical Benchmark Evaluation (Med-PaLM 2 / AMIE)",
      "difficulty": 3
    },
    {
      "id": 1555,
      "section": "Section 47 — Domain-Specific AI Architecture (Robotics, Bio, Finance & Software Agents)",
      "question": "FDA SaMD Regulations & Good Machine Learning Practice (GMLP)",
      "difficulty": 3
    },
    {
      "id": 1556,
      "section": "Section 47 — Domain-Specific AI Architecture (Robotics, Bio, Finance & Software Agents)",
      "question": "HIPAA Compliance, Privacy-Preserving AI, and Zero-Retention API Architecture",
      "difficulty": 2
    },
    {
      "id": 1557,
      "section": "Section 47 — Domain-Specific AI Architecture (Robotics, Bio, Finance & Software Agents)",
      "question": "Algorithmic Bias & Clinical Safety Drift Across Healthcare Networks",
      "difficulty": 2
    },
    {
      "id": 1558,
      "section": "Section 47 — Domain-Specific AI Architecture (Robotics, Bio, Finance & Software Agents)",
      "question": "Deep Limit Order Book (LOB) Modeling & Order Flow Imbalance (OFI)",
      "difficulty": 3
    },
    {
      "id": 1559,
      "section": "Section 47 — Domain-Specific AI Architecture (Robotics, Bio, Finance & Software Agents)",
      "question": "Ultra-Low Latency Inference Execution under Sub-10-Microsecond SLAs",
      "difficulty": 3
    },
    {
      "id": 1560,
      "section": "Section 47 — Domain-Specific AI Architecture (Robotics, Bio, Finance & Software Agents)",
      "question": "Microstructure Noise, Market Regime Shift, and Online Model Adaptation",
      "difficulty": 2
    },
    {
      "id": 1561,
      "section": "Section 47 — Domain-Specific AI Architecture (Robotics, Bio, Finance & Software Agents)",
      "question": "Repository-Level Workspace Indexing: AST, CPG, and Hybrid RAG",
      "difficulty": 3
    },
    {
      "id": 1562,
      "section": "Section 47 — Domain-Specific AI Architecture (Robotics, Bio, Finance & Software Agents)",
      "question": "Autonomous Agent Patch Generation & Execution Harness on SWE-bench",
      "difficulty": 3
    },
    {
      "id": 1563,
      "section": "Section 47 — Domain-Specific AI Architecture (Robotics, Bio, Finance & Software Agents)",
      "question": "SWE-bench Benchmarking Metrics, Data Leakage, and Test Generation",
      "difficulty": 2
    },
    {
      "id": 1564,
      "section": "Section 47 — Domain-Specific AI Architecture (Robotics, Bio, Finance & Software Agents)",
      "question": "Deterministic Tool Execution & Patching State Management in Coding Agents",
      "difficulty": 2
    },
    {
      "id": 1565,
      "section": "Section 47 — Domain-Specific AI Architecture (Robotics, Bio, Finance & Software Agents)",
      "question": "AlphaGenome & Genomic Foundation Models for Non-Coding Variant Prediction",
      "difficulty": 3
    },
    {
      "id": 1566,
      "section": "Section 47 — Domain-Specific AI Architecture (Robotics, Bio, Finance & Software Agents)",
      "question": "Ensembl VEP & ACMG/AMP Clinical Variant Classification Pipelines",
      "difficulty": 2
    },
    {
      "id": 1567,
      "section": "Section 47 — Domain-Specific AI Architecture (Robotics, Bio, Finance & Software Agents)",
      "question": "Splicing Disruption & Linkage Disequilibrium in Non-Coding Variant Analysis",
      "difficulty": 3
    },
    {
      "id": 1568,
      "section": "Section 47 — Domain-Specific AI Architecture (Robotics, Bio, Finance & Software Agents)",
      "question": "AI-Driven Drug Discovery: ChEMBL Integration & 3D GNN Affinity Modeling",
      "difficulty": 3
    },
    {
      "id": 1569,
      "section": "Section 47 — Domain-Specific AI Architecture (Robotics, Bio, Finance & Software Agents)",
      "question": "De Novo Generative Molecular Optimization & ADMET Property Prediction",
      "difficulty": 2
    },
    {
      "id": 1570,
      "section": "Section 47 — Domain-Specific AI Architecture (Robotics, Bio, Finance & Software Agents)",
      "question": "Causal Market Simulation & Counterfactual Backtesting for Trading Strategies",
      "difficulty": 3
    },
    {
      "id": 1571,
      "section": "Section 47 — Domain-Specific AI Architecture (Robotics, Bio, Finance & Software Agents)",
      "question": "Multi-Agent Reinforcement Learning (MARL) for Market Making & Trade Execution",
      "difficulty": 2
    },
    {
      "id": 1572,
      "section": "Section 48 — Deep-Dive Agentic Frameworks, AI Gateway Architecture & Token Budget Engineering",
      "question": "LangGraph StateGraph Architecture, Reducers & Channel Reducer Mechanics",
      "difficulty": 3
    },
    {
      "id": 1573,
      "section": "Section 48 — Deep-Dive Agentic Frameworks, AI Gateway Architecture & Token Budget Engineering",
      "question": "LangGraph Dynamic Routing, Conditional Edges & State Control Flow",
      "difficulty": 2
    },
    {
      "id": 1574,
      "section": "Section 48 — Deep-Dive Agentic Frameworks, AI Gateway Architecture & Token Budget Engineering",
      "question": "LangGraph State Persistence, Checkpointing & Human-in-the-Loop (HITL) Interruption",
      "difficulty": 3
    },
    {
      "id": 1575,
      "section": "Section 48 — Deep-Dive Agentic Frameworks, AI Gateway Architecture & Token Budget Engineering",
      "question": "CrewAI Framework Mechanics: Task Execution, Role Definition & Process Delegation",
      "difficulty": 2
    },
    {
      "id": 1576,
      "section": "Section 48 — Deep-Dive Agentic Frameworks, AI Gateway Architecture & Token Budget Engineering",
      "question": "CrewAI Manager Delegation Loops, Communication Protocols & Sub-Task Orchestration",
      "difficulty": 3
    },
    {
      "id": 1577,
      "section": "Section 48 — Deep-Dive Agentic Frameworks, AI Gateway Architecture & Token Budget Engineering",
      "question": "CrewAI Multi-Tiered Memory Systems: Short-Term, Long-Term & Entity Memory",
      "difficulty": 3
    },
    {
      "id": 1578,
      "section": "Section 48 — Deep-Dive Agentic Frameworks, AI Gateway Architecture & Token Budget Engineering",
      "question": "Microsoft AutoGen Architecture: ConversableAgent Message Handlers & Execution Routing",
      "difficulty": 2
    },
    {
      "id": 1579,
      "section": "Section 48 — Deep-Dive Agentic Frameworks, AI Gateway Architecture & Token Budget Engineering",
      "question": "AutoGen Multi-Agent GroupChat & Speaker Selection Algorithms",
      "difficulty": 3
    },
    {
      "id": 1580,
      "section": "Section 48 — Deep-Dive Agentic Frameworks, AI Gateway Architecture & Token Budget Engineering",
      "question": "AutoGen Sandboxed Code Execution Engine & Security Isolation",
      "difficulty": 3
    },
    {
      "id": 1581,
      "section": "Section 48 — Deep-Dive Agentic Frameworks, AI Gateway Architecture & Token Budget Engineering",
      "question": "LlamaIndex Event-Driven Workflows: `@step` Decorators & Event-Based Async Pipelines",
      "difficulty": 2
    },
    {
      "id": 1582,
      "section": "Section 48 — Deep-Dive Agentic Frameworks, AI Gateway Architecture & Token Budget Engineering",
      "question": "LlamaIndex Workflow Integration: Building Custom ReAct and Function Calling Agents",
      "difficulty": 3
    },
    {
      "id": 1583,
      "section": "Section 48 — Deep-Dive Agentic Frameworks, AI Gateway Architecture & Token Budget Engineering",
      "question": "Microsoft Semantic Kernel Architecture: Native Plugins, Prompt Plugins & Kernel Arguments",
      "difficulty": 2
    },
    {
      "id": 1584,
      "section": "Section 48 — Deep-Dive Agentic Frameworks, AI Gateway Architecture & Token Budget Engineering",
      "question": "Semantic Kernel Automated Planning: Sequential & Stepwise Planner Mechanics",
      "difficulty": 3
    },
    {
      "id": 1585,
      "section": "Section 48 — Deep-Dive Agentic Frameworks, AI Gateway Architecture & Token Budget Engineering",
      "question": "AI Gateway Semantic Caching Architecture: Vector Similarity Lookups & TTL Hygiene",
      "difficulty": 2
    },
    {
      "id": 1586,
      "section": "Section 48 — Deep-Dive Agentic Frameworks, AI Gateway Architecture & Token Budget Engineering",
      "question": "Semantic Caching Data Protection: PII Masking, Scrubbing & Multi-Tenant Isolation",
      "difficulty": 3
    },
    {
      "id": 1587,
      "section": "Section 48 — Deep-Dive Agentic Frameworks, AI Gateway Architecture & Token Budget Engineering",
      "question": "AI Gateway Multi-Cloud Load Balancing: Weighted Routing & Cloud Provider Failover",
      "difficulty": 2
    },
    {
      "id": 1588,
      "section": "Section 48 — Deep-Dive Agentic Frameworks, AI Gateway Architecture & Token Budget Engineering",
      "question": "AI Gateway Resiliency Patterns: Circuit Breakers, Probing & Exponential Backoff",
      "difficulty": 3
    },
    {
      "id": 1589,
      "section": "Section 48 — Deep-Dive Agentic Frameworks, AI Gateway Architecture & Token Budget Engineering",
      "question": "Enterprise Rate Limiting, Token Buckets & Cost Attribution at the Gateway",
      "difficulty": 2
    },
    {
      "id": 1590,
      "section": "Section 48 — Deep-Dive Agentic Frameworks, AI Gateway Architecture & Token Budget Engineering",
      "question": "Dynamic Context Token Budget Allocator: Sliding Window Partitioning Engine",
      "difficulty": 3
    },
    {
      "id": 1591,
      "section": "Section 48 — Deep-Dive Agentic Frameworks, AI Gateway Architecture & Token Budget Engineering",
      "question": "Context Window Compaction & Priority-Based Eviction Algorithms",
      "difficulty": 2
    },
    {
      "id": 1592,
      "section": "Section 48 — Deep-Dive Agentic Frameworks, AI Gateway Architecture & Token Budget Engineering",
      "question": "Prompt Compression Mechanics: LLMLingua Perplexity-Based Pruning",
      "difficulty": 3
    },
    {
      "id": 1593,
      "section": "Section 48 — Deep-Dive Agentic Frameworks, AI Gateway Architecture & Token Budget Engineering",
      "question": "Selective Context & Information Entropy Pruning Mechanics",
      "difficulty": 2
    },
    {
      "id": 1594,
      "section": "Section 48 — Deep-Dive Agentic Frameworks, AI Gateway Architecture & Token Budget Engineering",
      "question": "Immutable Agent Action Ledger: Hash-Chained Event Logs for Enterprise Auditing",
      "difficulty": 3
    },
    {
      "id": 1595,
      "section": "Section 48 — Deep-Dive Agentic Frameworks, AI Gateway Architecture & Token Budget Engineering",
      "question": "Merkle-Tree Based Compliance Verification for Distributed Multi-Agent Systems",
      "difficulty": 3
    },
    {
      "id": 1596,
      "section": "Section 48 — Deep-Dive Agentic Frameworks, AI Gateway Architecture & Token Budget Engineering",
      "question": "Comprehensive Architecture Synthesis: Enterprise AI Gateway & Multi-Agent Framework Orchestration",
      "difficulty": 3
    },
    {
      "id": 1597,
      "section": "Section 49 — Enterprise Cloud AI Deployment Architectures (AWS, Azure & GCP)",
      "question": "AWS Amazon Bedrock Provisioned Throughput vs On-Demand Allocation & Quota Management",
      "difficulty": 2
    },
    {
      "id": 1598,
      "section": "Section 49 — Enterprise Cloud AI Deployment Architectures (AWS, Azure & GCP)",
      "question": "AWS Bedrock Guardrails Architecture, Content Filtering & VPC PrivateLink Endpoints",
      "difficulty": 3
    },
    {
      "id": 1599,
      "section": "Section 49 — Enterprise Cloud AI Deployment Architectures (AWS, Azure & GCP)",
      "question": "AWS Bedrock Custom Model Import (CMI) & Fine-Tuned Model Deployment",
      "difficulty": 2
    },
    {
      "id": 1600,
      "section": "Section 49 — Enterprise Cloud AI Deployment Architectures (AWS, Azure & GCP)",
      "question": "AWS SageMaker Real-Time & Async Inference: Multi-Model Endpoints (MME) & Dynamic GPU Loading",
      "difficulty": 3
    },
    {
      "id": 1601,
      "section": "Section 49 — Enterprise Cloud AI Deployment Architectures (AWS, Azure & GCP)",
      "question": "AWS SageMaker GPU Auto-Scaling & Deep Learning Containers (DLC)",
      "difficulty": 2
    },
    {
      "id": 1602,
      "section": "Section 49 — Enterprise Cloud AI Deployment Architectures (AWS, Azure & GCP)",
      "question": "AWS Custom Silicon Architecture: AWS Neuron SDK Toolchain & NeuronCore Pipeline Parallelism",
      "difficulty": 3
    },
    {
      "id": 1603,
      "section": "Section 49 — Enterprise Cloud AI Deployment Architectures (AWS, Azure & GCP)",
      "question": "AWS Inferentia2 vs NVIDIA H100/A10G Benchmark & Latency-Cost Optimization",
      "difficulty": 3
    },
    {
      "id": 1604,
      "section": "Section 49 — Enterprise Cloud AI Deployment Architectures (AWS, Azure & GCP)",
      "question": "AWS EKS for GenAI: Karpenter Node Autoscaling & GPU Instance Provisioning",
      "difficulty": 2
    },
    {
      "id": 1605,
      "section": "Section 49 — Enterprise Cloud AI Deployment Architectures (AWS, Azure & GCP)",
      "question": "AWS EKS Multi-Node Distributed Training & Ray Orchestration via KubeRay & Service Mesh",
      "difficulty": 3
    },
    {
      "id": 1606,
      "section": "Section 49 — Enterprise Cloud AI Deployment Architectures (AWS, Azure & GCP)",
      "question": "Azure OpenAI Service Capacity Planning: PTU vs PAYG Architecture & Token Allocation",
      "difficulty": 2
    },
    {
      "id": 1607,
      "section": "Section 49 — Enterprise Cloud AI Deployment Architectures (AWS, Azure & GCP)",
      "question": "Azure Managed Identity Zero-Trust Authentication & Private Endpoint Network Topology",
      "difficulty": 3
    },
    {
      "id": 1608,
      "section": "Section 49 — Enterprise Cloud AI Deployment Architectures (AWS, Azure & GCP)",
      "question": "Azure OpenAI Regional Availability Failover & Multi-Region Gateway Design",
      "difficulty": 3
    },
    {
      "id": 1609,
      "section": "Section 49 — Enterprise Cloud AI Deployment Architectures (AWS, Azure & GCP)",
      "question": "Azure Machine Learning (AML) Managed Endpoints: vLLM Containers & Blue/Green Deployments",
      "difficulty": 2
    },
    {
      "id": 1610,
      "section": "Section 49 — Enterprise Cloud AI Deployment Architectures (AWS, Azure & GCP)",
      "question": "Azure Kubernetes Service (AKS) GenAI Scaling: KEDA Queue Depth & TPOT Latency Metrics",
      "difficulty": 3
    },
    {
      "id": 1611,
      "section": "Section 49 — Enterprise Cloud AI Deployment Architectures (AWS, Azure & GCP)",
      "question": "Enterprise Azure RAG Stack: Azure OpenAI + AI Search + Cosmos DB + APIM AI Gateway",
      "difficulty": 3
    },
    {
      "id": 1612,
      "section": "Section 49 — Enterprise Cloud AI Deployment Architectures (AWS, Azure & GCP)",
      "question": "GCP Vertex AI Model Garden & Endpoint Serving: vLLM on G2 & A3 Mega Instances",
      "difficulty": 2
    },
    {
      "id": 1613,
      "section": "Section 49 — Enterprise Cloud AI Deployment Architectures (AWS, Azure & GCP)",
      "question": "GCP TPU v5e/v6 Trillium Slice Serving & Vertex AI Prediction SLA Monitoring",
      "difficulty": 3
    },
    {
      "id": 1614,
      "section": "Section 49 — Enterprise Cloud AI Deployment Architectures (AWS, Azure & GCP)",
      "question": "GCP Cloud Run GPU Serverless Inference: L4 GPU Containerization & VPC Service Controls",
      "difficulty": 3
    },
    {
      "id": 1615,
      "section": "Section 49 — Enterprise Cloud AI Deployment Architectures (AWS, Azure & GCP)",
      "question": "GCP Kubernetes Engine (GKE) for AI: GPU Auto-Provisioning & TPU Pod Slice Scheduling",
      "difficulty": 2
    },
    {
      "id": 1616,
      "section": "Section 49 — Enterprise Cloud AI Deployment Architectures (AWS, Azure & GCP)",
      "question": "GCP Enterprise RAG & Vector Data Stack: Vertex AI Search, BigQuery ML & AlloyDB pgvector",
      "difficulty": 3
    },
    {
      "id": 1617,
      "section": "Section 49 — Enterprise Cloud AI Deployment Architectures (AWS, Azure & GCP)",
      "question": "Multi-Cloud IaC: Terraform Modules for Cross-Cloud LLM Gateway Infrastructure",
      "difficulty": 3
    },
    {
      "id": 1618,
      "section": "Section 49 — Enterprise Cloud AI Deployment Architectures (AWS, Azure & GCP)",
      "question": "Multi-Cloud IaC: Pulumi Infrastructure-as-Code for GenAI Orchestration",
      "difficulty": 2
    },
    {
      "id": 1619,
      "section": "Section 49 — Enterprise Cloud AI Deployment Architectures (AWS, Azure & GCP)",
      "question": "FinOps & Cloud AI Cost Governance: Spot/Preemptible GPUs vs CUDs & Savings Plans",
      "difficulty": 3
    },
    {
      "id": 1620,
      "section": "Section 49 — Enterprise Cloud AI Deployment Architectures (AWS, Azure & GCP)",
      "question": "GPU Utilization Telemetry: Prometheus + DCGM Exporter & Idle Instance Auto-Termination",
      "difficulty": 2
    },
    {
      "id": 1621,
      "section": "Section 49 — Enterprise Cloud AI Deployment Architectures (AWS, Azure & GCP)",
      "question": "End-to-End Enterprise Multi-Cloud AI Architecture Blueprint",
      "difficulty": 3
    },
    {
      "id": 1622,
      "section": "Section 50 — Production Incident Triage & Live Debugging",
      "question": "Your LLM application suddenly returns HTTP 429s in production. Before assuming \"too many requests,\" what distinct limits could actually be firing, and how do you identify which one in the first five minutes?",
      "difficulty": 3
    },
    {
      "id": 1623,
      "section": "Section 50 — Production Incident Triage & Live Debugging",
      "question": "RPM sits at 40% of quota but you are still getting 429s. Walk through what you check next and what each signal rules in or out.",
      "difficulty": 3
    },
    {
      "id": 1624,
      "section": "Section 50 — Production Incident Triage & Live Debugging",
      "question": "Retrieval starts returning 20 chunks instead of 5 after a config change. Request volume is unchanged. Trace the full blast radius of that single change through rate limits, latency, cost and answer quality.",
      "difficulty": 3
    },
    {
      "id": 1625,
      "section": "Section 50 — Production Incident Triage & Live Debugging",
      "question": "Application traffic looks flat but downstream model calls have tripled. What class of change causes this, and how do you prove it from telemetry rather than guessing?",
      "difficulty": 3
    },
    {
      "id": 1626,
      "section": "Section 50 — Production Incident Triage & Live Debugging",
      "question": "Your service retries every 429 with immediate retry. Explain the failure mode this creates, why it is self-reinforcing, and the specific retry policy you would replace it with.",
      "difficulty": 3
    },
    {
      "id": 1627,
      "section": "Section 50 — Production Incident Triage & Live Debugging",
      "question": "Average RPM looks healthy all day yet you get 429 bursts at unpredictable moments. What metric is hiding the problem, and how do you instrument for it?",
      "difficulty": 2
    },
    {
      "id": 1628,
      "section": "Section 50 — Production Incident Triage & Live Debugging",
      "question": "A provider returns 429 with no `Retry-After` header and a vague error body. How do you build a client that behaves correctly under this uncertainty without hammering the provider?",
      "difficulty": 2
    },
    {
      "id": 1629,
      "section": "Section 50 — Production Incident Triage & Live Debugging",
      "question": "Distinguish a rate limit you are causing from a provider-side capacity incident. What evidence separates the two, and how does your response differ?",
      "difficulty": 3
    },
    {
      "id": 1630,
      "section": "Section 50 — Production Incident Triage & Live Debugging",
      "question": "p50 latency is unchanged but p99 has doubled overnight. Enumerate the candidate causes specific to LLM serving and the order you would eliminate them.",
      "difficulty": 3
    },
    {
      "id": 1631,
      "section": "Section 50 — Production Incident Triage & Live Debugging",
      "question": "Time to first token is fine but tokens per second has degraded. What does that split tell you about where the bottleneck is?",
      "difficulty": 3
    },
    {
      "id": 1632,
      "section": "Section 50 — Production Incident Triage & Live Debugging",
      "question": "Your monthly LLM spend doubled with flat user numbers. Build the diagnostic tree that gets you from the bill to the responsible code path.",
      "difficulty": 3
    },
    {
      "id": 1633,
      "section": "Section 50 — Production Incident Triage & Live Debugging",
      "question": "Users report the assistant \"got worse this week.\" No deployment went out and no model version changed. What can still have changed, and how do you confirm it?",
      "difficulty": 3
    },
    {
      "id": 1634,
      "section": "Section 50 — Production Incident Triage & Live Debugging",
      "question": "Your RAG system's answers degraded but retrieval metrics look unchanged. Where do you look, and why can retrieval metrics stay flat while quality falls?",
      "difficulty": 3
    },
    {
      "id": 1635,
      "section": "Section 50 — Production Incident Triage & Live Debugging",
      "question": "An agent that normally completes a task in 6 steps is now taking 40 and sometimes never finishing. Diagnose systematically rather than by raising the step cap.",
      "difficulty": 3
    },
    {
      "id": 1636,
      "section": "Section 50 — Production Incident Triage & Live Debugging",
      "question": "Streaming responses intermittently truncate mid-sentence for a subset of users. Work through the layers where this can originate.",
      "difficulty": 2
    },
    {
      "id": 1637,
      "section": "Section 50 — Production Incident Triage & Live Debugging",
      "question": "Your self-hosted vLLM deployment starts OOM-ing under traffic it previously handled. What changed characteristics of the workload would explain it, and what do you check first?",
      "difficulty": 3
    },
    {
      "id": 1638,
      "section": "Section 50 — Production Incident Triage & Live Debugging",
      "question": "Vector search recall has quietly dropped over three months with no code change. Explain the mechanism and how you would have detected it earlier.",
      "difficulty": 3
    },
    {
      "id": 1639,
      "section": "Section 50 — Production Incident Triage & Live Debugging",
      "question": "A single tenant's traffic is degrading latency for everyone else on shared infrastructure. Identify the isolation failure and the controls that should have prevented it.",
      "difficulty": 3
    },
    {
      "id": 1640,
      "section": "Section 50 — Production Incident Triage & Live Debugging",
      "question": "Your eval suite is green but users are complaining. Reconcile the contradiction and describe what you change so the suite stops lying to you.",
      "difficulty": 3
    },
    {
      "id": 1641,
      "section": "Section 50 — Production Incident Triage & Live Debugging",
      "question": "A prompt change shipped four hours ago and error rates are climbing slowly rather than spiking. Why is gradual degradation harder to attribute, and how do you handle it?",
      "difficulty": 2
    },
    {
      "id": 1642,
      "section": "Section 50 — Production Incident Triage & Live Debugging",
      "question": "Guardrail false-positive rate jumped after a model version update you did not initiate. Explain how a provider-side change surfaces this way and what your standing defence is.",
      "difficulty": 3
    },
    {
      "id": 1643,
      "section": "Section 50 — Production Incident Triage & Live Debugging",
      "question": "Tool calls are failing intermittently with malformed arguments, but only for some tools. Diagnose whether the cause is the model, the schema, or the input distribution.",
      "difficulty": 3
    },
    {
      "id": 1644,
      "section": "Section 50 — Production Incident Triage & Live Debugging",
      "question": "Your agent's cost per successful task has risen while its success rate stayed flat. What is happening, and why is success rate alone a misleading health metric?",
      "difficulty": 3
    },
    {
      "id": 1645,
      "section": "Section 50 — Production Incident Triage & Live Debugging",
      "question": "Structure the first ten minutes of an AI incident: what you check, what you communicate, and what you deliberately do not do yet.",
      "difficulty": 3
    },
    {
      "id": 1646,
      "section": "Section 50 — Production Incident Triage & Live Debugging",
      "question": "Write the postmortem for an LLM incident where the root cause was \"the model returned something unexpected.\" Explain why that phrasing is unacceptable and what a real root cause statement looks like.",
      "difficulty": 3
    },
    {
      "id": 1647,
      "section": "Section 51 — Multi-Turn Interviewer Drills",
      "question": "**429s in production.** Interviewer: \"Your LLM application suddenly returns 429s. What do you check?\" — then follows up four times as each hypothesis is eliminated.",
      "difficulty": 3
    },
    {
      "id": 1648,
      "section": "Section 51 — Multi-Turn Interviewer Drills",
      "question": "**RAG answers are wrong.** Interviewer: \"Users say the assistant cites the right document but gives the wrong answer.\" — drills into chunking, grounding and the eval blind spot.",
      "difficulty": 3
    },
    {
      "id": 1649,
      "section": "Section 51 — Multi-Turn Interviewer Drills",
      "question": "**The cost conversation.** Interviewer: \"Your feature costs $180k/month. The CFO wants it at $60k without quality loss. Where do you start?\" — pushes on each lever's real limit.",
      "difficulty": 3
    },
    {
      "id": 1650,
      "section": "Section 51 — Multi-Turn Interviewer Drills",
      "question": "**Agent went rogue.** Interviewer: \"An agent deleted production data. Walk me through what failed.\" — escalates from the immediate bug to the governance gap.",
      "difficulty": 3
    },
    {
      "id": 1651,
      "section": "Section 51 — Multi-Turn Interviewer Drills",
      "question": "**Fine-tune or not.** Interviewer: \"The team wants to fine-tune. Convince me it's the wrong call — or the right one.\" — tests whether you argue from evidence or fashion.",
      "difficulty": 3
    },
    {
      "id": 1652,
      "section": "Section 51 — Multi-Turn Interviewer Drills",
      "question": "**Latency budget.** Interviewer: \"Product wants sub-second responses for a RAG feature currently at 4s. Is that achievable?\" — forces honest scoping rather than agreement.",
      "difficulty": 3
    },
    {
      "id": 1653,
      "section": "Section 51 — Multi-Turn Interviewer Drills",
      "question": "**The eval is lying.** Interviewer: \"Your eval scores went up, your users are unhappier. Explain.\" — drills into judge calibration and set drift.",
      "difficulty": 3
    },
    {
      "id": 1654,
      "section": "Section 51 — Multi-Turn Interviewer Drills",
      "question": "**Provider deprecation.** Interviewer: \"Your provider deprecates the model you depend on in 30 days. Go.\" — tests incident-grade planning under a hard deadline.",
      "difficulty": 3
    },
    {
      "id": 1655,
      "section": "Section 51 — Multi-Turn Interviewer Drills",
      "question": "**Hallucination in a regulated context.** Interviewer: \"Your system gave a customer incorrect financial guidance. What now?\" — escalates through containment, disclosure and prevention.",
      "difficulty": 3
    },
    {
      "id": 1656,
      "section": "Section 51 — Multi-Turn Interviewer Drills",
      "question": "**Scaling a prototype.** Interviewer: \"The demo works. It goes to 50,000 users on Monday. What breaks first?\" — tests whether you can predict failure order.",
      "difficulty": 3
    },
    {
      "id": 1657,
      "section": "Section 51 — Multi-Turn Interviewer Drills",
      "question": "**The vector database question.** Interviewer: \"Why did you choose a dedicated vector DB over pgvector?\" — pushes until you either justify or concede.",
      "difficulty": 3
    },
    {
      "id": 1658,
      "section": "Section 51 — Multi-Turn Interviewer Drills",
      "question": "**Multi-agent scepticism.** Interviewer: \"Why not just use one agent with more tools?\" — tests whether you can defend or abandon multi-agent complexity.",
      "difficulty": 3
    },
    {
      "id": 1659,
      "section": "Section 51 — Multi-Turn Interviewer Drills",
      "question": "**Prompt injection in an enterprise deployment.** Interviewer: \"A user got your agent to email them another customer's data. How?\" — traces the exploit chain and the missing controls.",
      "difficulty": 3
    },
    {
      "id": 1660,
      "section": "Section 51 — Multi-Turn Interviewer Drills",
      "question": "**The disagreement.** Interviewer asserts something technically wrong and holds their position. Tests whether you fold, escalate badly, or disagree well.",
      "difficulty": 3
    },
    {
      "id": 1661,
      "section": "Section 51 — Multi-Turn Interviewer Drills",
      "question": "**Explaining to the board.** Interviewer: \"Explain in two minutes, no jargon, why the AI programme needs another $4M.\" — tests translation, not technical depth.",
      "difficulty": 3
    },
    {
      "id": 1662,
      "section": "Section 52 — Spot the Flaw: Design & Code Critique",
      "question": "A team caches LLM responses keyed on `hash(user_prompt)` in a shared Redis, TTL 24h, to cut cost on a personalised assistant. What's wrong?",
      "difficulty": 2
    },
    {
      "id": 1663,
      "section": "Section 52 — Spot the Flaw: Design & Code Critique",
      "question": "A RAG pipeline embeds the user's raw question, retrieves top-50 by cosine similarity, concatenates all 50 chunks, and sends them with the question. Critique it.",
      "difficulty": 2
    },
    {
      "id": 1664,
      "section": "Section 52 — Spot the Flaw: Design & Code Critique",
      "question": "An agent's retry logic: `for attempt in range(5): try: return call_llm(p) except: continue`. List every problem.",
      "difficulty": 3
    },
    {
      "id": 1665,
      "section": "Section 52 — Spot the Flaw: Design & Code Critique",
      "question": "A team enforces JSON output by appending \"Respond only in valid JSON\" to the prompt and calling `json.loads()` on the result. What breaks, and what should they do instead?",
      "difficulty": 2
    },
    {
      "id": 1666,
      "section": "Section 52 — Spot the Flaw: Design & Code Critique",
      "question": "Guardrails are implemented as a single output classifier that blocks unsafe responses. The team calls this \"defence in depth.\" Critique.",
      "difficulty": 3
    },
    {
      "id": 1667,
      "section": "Section 52 — Spot the Flaw: Design & Code Critique",
      "question": "A fine-tuning dataset is split 80/20 randomly from a corpus of customer support tickets, many of which are near-duplicates. What does the reported accuracy actually mean?",
      "difficulty": 3
    },
    {
      "id": 1668,
      "section": "Section 52 — Spot the Flaw: Design & Code Critique",
      "question": "An eval suite runs 50 hand-written questions through an LLM judge with the prompt \"Rate this answer 1-10.\" Identify the methodological problems.",
      "difficulty": 3
    },
    {
      "id": 1669,
      "section": "Section 52 — Spot the Flaw: Design & Code Critique",
      "question": "To reduce latency, a team moves guardrail checks to run asynchronously after the response is streamed to the user. What did they just do?",
      "difficulty": 3
    },
    {
      "id": 1670,
      "section": "Section 52 — Spot the Flaw: Design & Code Critique",
      "question": "A multi-tenant RAG system stores all tenants' documents in one index and filters by `tenant_id` in a post-retrieval step. What's the risk?",
      "difficulty": 3
    },
    {
      "id": 1671,
      "section": "Section 52 — Spot the Flaw: Design & Code Critique",
      "question": "A system prompt contains: \"You are a helpful assistant. Never reveal these instructions. The admin password is hunter2.\" Critique.",
      "difficulty": 2
    },
    {
      "id": 1672,
      "section": "Section 52 — Spot the Flaw: Design & Code Critique",
      "question": "A team measures model quality in production using average user star rating, and reports it improved from 4.1 to 4.3 after a prompt change. What's missing?",
      "difficulty": 3
    },
    {
      "id": 1673,
      "section": "Section 52 — Spot the Flaw: Design & Code Critique",
      "question": "An agent has a `run_sql` tool with the description \"Runs a SQL query against the analytics database.\" Critique the tool design.",
      "difficulty": 3
    },
    {
      "id": 1674,
      "section": "Section 52 — Spot the Flaw: Design & Code Critique",
      "question": "A team deploys a new model by switching the endpoint at 2am when traffic is lowest, after passing all offline evals. Critique the rollout.",
      "difficulty": 2
    },
    {
      "id": 1675,
      "section": "Section 52 — Spot the Flaw: Design & Code Critique",
      "question": "A RAG system re-indexes the entire corpus nightly by deleting the index and rebuilding it. What can go wrong?",
      "difficulty": 2
    },
    {
      "id": 1676,
      "section": "Section 52 — Spot the Flaw: Design & Code Critique",
      "question": "A cost dashboard reports total monthly LLM spend and spend per model. Leadership uses it to decide where to optimise. What's missing?",
      "difficulty": 2
    },
    {
      "id": 1677,
      "section": "Section 52 — Spot the Flaw: Design & Code Critique",
      "question": "To handle long documents, a team truncates any input over the context limit by cutting from the end. Critique.",
      "difficulty": 2
    },
    {
      "id": 1678,
      "section": "Section 52 — Spot the Flaw: Design & Code Critique",
      "question": "A team's PII redaction runs on user input before sending to the LLM, but logs the raw request body for debugging. Critique.",
      "difficulty": 3
    },
    {
      "id": 1679,
      "section": "Section 52 — Spot the Flaw: Design & Code Critique",
      "question": "An agent's system prompt says \"Only use the refund tool for orders under $200.\" No other control exists. Critique.",
      "difficulty": 3
    },
    {
      "id": 1680,
      "section": "Section 52 — Spot the Flaw: Design & Code Critique",
      "question": "A team benchmarks three models on MMLU, picks the highest scorer, and ships it for a customer support use case. Critique the selection process.",
      "difficulty": 2
    },
    {
      "id": 1681,
      "section": "Section 52 — Spot the Flaw: Design & Code Critique",
      "question": "Load testing is done by sending 1,000 identical requests concurrently and measuring throughput. Why is this misleading for LLM serving?",
      "difficulty": 3
    },
    {
      "id": 1682,
      "section": "Section 53 — Estimation, Capacity & Cost Arithmetic",
      "question": "Estimate the monthly API cost of a support assistant: 50,000 conversations/month, 6 turns each, 2,000 input tokens and 300 output tokens per turn, at $3/M input and $15/M output.",
      "difficulty": 2
    },
    {
      "id": 1683,
      "section": "Section 53 — Estimation, Capacity & Cost Arithmetic",
      "question": "How much GPU memory does a 70B parameter model need for inference at FP16, and what changes at INT8 and INT4?",
      "difficulty": 2
    },
    {
      "id": 1684,
      "section": "Section 53 — Estimation, Capacity & Cost Arithmetic",
      "question": "Estimate the KV cache size per request for a 70B model with 80 layers, 64 heads, head dimension 128, at 8,000 tokens of context in FP16. What does that imply for concurrency on an 80GB GPU?",
      "difficulty": 3
    },
    {
      "id": 1685,
      "section": "Section 53 — Estimation, Capacity & Cost Arithmetic",
      "question": "You need to serve 100 requests/second with an average of 500 output tokens. If one H100 delivers roughly 2,500 output tokens/second for your model, how many GPUs do you need, and what's wrong with that calculation?",
      "difficulty": 3
    },
    {
      "id": 1686,
      "section": "Section 53 — Estimation, Capacity & Cost Arithmetic",
      "question": "Estimate the storage and memory footprint of a vector index: 10 million chunks, 1,536-dimensional float32 embeddings, HNSW.",
      "difficulty": 2
    },
    {
      "id": 1687,
      "section": "Section 53 — Estimation, Capacity & Cost Arithmetic",
      "question": "A RAG feature adds 6,000 tokens of retrieved context to every request. At 2 million requests/month and $3/M input tokens, what does retrieval breadth cost you annually?",
      "difficulty": 2
    },
    {
      "id": 1688,
      "section": "Section 53 — Estimation, Capacity & Cost Arithmetic",
      "question": "Estimate the cost and wall-clock time to fine-tune a 7B model with LoRA on 50,000 examples of ~1,000 tokens each, on 8×A100.",
      "difficulty": 3
    },
    {
      "id": 1689,
      "section": "Section 53 — Estimation, Capacity & Cost Arithmetic",
      "question": "Your agent averages 12 LLM calls per task at 3,000 input and 500 output tokens per call. Compute cost per task and per 10,000 tasks at $1/M input, $5/M output.",
      "difficulty": 2
    },
    {
      "id": 1690,
      "section": "Section 53 — Estimation, Capacity & Cost Arithmetic",
      "question": "Estimate the embedding cost and time to index a 5-million-document corpus averaging 4 chunks per document at $0.02/M tokens and 400 tokens per chunk.",
      "difficulty": 2
    },
    {
      "id": 1691,
      "section": "Section 53 — Estimation, Capacity & Cost Arithmetic",
      "question": "A latency budget is 2 seconds end to end. Allocate it across retrieval, re-ranking, prefill and decode for a 500-token answer, and state what you'd cut first if you missed.",
      "difficulty": 3
    },
    {
      "id": 1692,
      "section": "Section 53 — Estimation, Capacity & Cost Arithmetic",
      "question": "Estimate how many concurrent users a single replica can serve if each request takes 3 seconds and users send one request every 30 seconds.",
      "difficulty": 2
    },
    {
      "id": 1693,
      "section": "Section 53 — Estimation, Capacity & Cost Arithmetic",
      "question": "Semantic caching achieves a 40% hit rate. Quantify the cost saving and explain why the latency saving is larger than the cost saving in percentage terms.",
      "difficulty": 2
    },
    {
      "id": 1694,
      "section": "Section 53 — Estimation, Capacity & Cost Arithmetic",
      "question": "Your provider allows 200,000 TPM. Given 2,500 input and 400 output tokens per request, what request rate can you sustain, and what breaks the calculation?",
      "difficulty": 3
    },
    {
      "id": 1695,
      "section": "Section 53 — Estimation, Capacity & Cost Arithmetic",
      "question": "Estimate the annual cost difference between self-hosting a 13B model on reserved GPUs versus using a comparable hosted API at 20 million requests/month. State the crossover point.",
      "difficulty": 3
    },
    {
      "id": 1696,
      "section": "Section 53 — Estimation, Capacity & Cost Arithmetic",
      "question": "How much training compute (in FLOPs) does a 7B model on 2 trillion tokens require, and roughly how many GPU-hours is that?",
      "difficulty": 3
    },
    {
      "id": 1697,
      "section": "Section 53 — Estimation, Capacity & Cost Arithmetic",
      "question": "You have 200 hours of engineer time to cut LLM spend. Rank the levers by expected saving per engineer-hour and justify the ordering.",
      "difficulty": 3
    },
    {
      "id": 1698,
      "section": "Section 53 — Estimation, Capacity & Cost Arithmetic",
      "question": "Estimate the p99 impact of a 5% cold-start rate on an autoscaled deployment where cold start costs 8 seconds.",
      "difficulty": 3
    },
    {
      "id": 1699,
      "section": "Section 53 — Estimation, Capacity & Cost Arithmetic",
      "question": "A team wants 99.9% availability for an LLM feature whose sole provider offers 99.5%. Is that achievable, and what does it require?",
      "difficulty": 3
    },
    {
      "id": 1700,
      "section": "Section 53 — Estimation, Capacity & Cost Arithmetic",
      "question": "Estimate how many labelled examples you need to detect a 2% quality regression with reasonable confidence, and explain why most eval suites are underpowered.",
      "difficulty": 3
    },
    {
      "id": 1701,
      "section": "Section 53 — Estimation, Capacity & Cost Arithmetic",
      "question": "Your context window is 128k tokens. Estimate how many pages of a typical PDF that is, and why the practical limit is far lower.",
      "difficulty": 2
    },
    {
      "id": 1702,
      "section": "Section 54 — Executive & Stakeholder Communication",
      "question": "Explain what a large language model is to a board with no technical background, in under 90 seconds, without using the words model, token, training or neural.",
      "difficulty": 2
    },
    {
      "id": 1703,
      "section": "Section 54 — Executive & Stakeholder Communication",
      "question": "Your CEO read that a competitor \"replaced 30% of engineering with AI\" and wants the same. How do you respond in the meeting?",
      "difficulty": 3
    },
    {
      "id": 1704,
      "section": "Section 54 — Executive & Stakeholder Communication",
      "question": "Explain to a CFO why AI costs are variable and usage-driven rather than a fixed licence, and what that means for budgeting.",
      "difficulty": 3
    },
    {
      "id": 1705,
      "section": "Section 54 — Executive & Stakeholder Communication",
      "question": "A product manager asks why the AI feature \"sometimes gets it wrong\" and wants it fixed to 100%. How do you set expectations without sounding defeatist?",
      "difficulty": 3
    },
    {
      "id": 1706,
      "section": "Section 54 — Executive & Stakeholder Communication",
      "question": "Write the three-sentence status update for a red-status AI project, addressed to an executive sponsor.",
      "difficulty": 2
    },
    {
      "id": 1707,
      "section": "Section 54 — Executive & Stakeholder Communication",
      "question": "Explain hallucination to a legal team assessing liability, in terms that are useful for their risk assessment rather than technically complete.",
      "difficulty": 3
    },
    {
      "id": 1708,
      "section": "Section 54 — Executive & Stakeholder Communication",
      "question": "Your team wants to spend a quarter on evaluation infrastructure with no user-visible output. Justify it to a product leader who is measured on shipped features.",
      "difficulty": 3
    },
    {
      "id": 1709,
      "section": "Section 54 — Executive & Stakeholder Communication",
      "question": "Explain to a customer's security team why sending their data to a third-party model provider is or isn't acceptable, without hiding behind certifications.",
      "difficulty": 3
    },
    {
      "id": 1710,
      "section": "Section 54 — Executive & Stakeholder Communication",
      "question": "How do you tell an executive sponsor that the AI project they championed should be cancelled?",
      "difficulty": 3
    },
    {
      "id": 1711,
      "section": "Section 54 — Executive & Stakeholder Communication",
      "question": "Explain the difference between a demo and a production system to a stakeholder who saw the demo work perfectly and can't understand the delay.",
      "difficulty": 3
    },
    {
      "id": 1712,
      "section": "Section 54 — Executive & Stakeholder Communication",
      "question": "A regulator asks how your system makes decisions. Structure your answer.",
      "difficulty": 3
    },
    {
      "id": 1713,
      "section": "Section 54 — Executive & Stakeholder Communication",
      "question": "Explain to a sales leader why you can't promise a customer a specific accuracy number in a contract.",
      "difficulty": 3
    },
    {
      "id": 1714,
      "section": "Section 54 — Executive & Stakeholder Communication",
      "question": "Your AI system caused a customer-visible incident. Draft the customer-facing communication.",
      "difficulty": 3
    },
    {
      "id": 1715,
      "section": "Section 54 — Executive & Stakeholder Communication",
      "question": "Explain to an engineering team why their elegant technical solution is being deprioritised for business reasons, without losing their trust.",
      "difficulty": 3
    },
    {
      "id": 1716,
      "section": "Section 54 — Executive & Stakeholder Communication",
      "question": "Present a build-versus-buy recommendation to an executive committee where the technically superior option is the one you're recommending against.",
      "difficulty": 3
    },
    {
      "id": 1717,
      "section": "Section 55 — Voice, Vision & Computer-Use Agents",
      "question": "Compare the cascaded ASR → LLM → TTS pipeline against a native speech-to-speech model. What does each win and lose?",
      "difficulty": 3
    },
    {
      "id": 1718,
      "section": "Section 55 — Voice, Vision & Computer-Use Agents",
      "question": "What is endpointing in a voice agent, and why is it the single hardest part of making conversation feel natural?",
      "difficulty": 3
    },
    {
      "id": 1719,
      "section": "Section 55 — Voice, Vision & Computer-Use Agents",
      "question": "Explain barge-in, and the acoustic echo problem it creates. How do you build it without headphones?",
      "difficulty": 3
    },
    {
      "id": 1720,
      "section": "Section 55 — Voice, Vision & Computer-Use Agents",
      "question": "Design the latency budget for a voice agent targeting sub-800ms perceived response. Where does the time actually go?",
      "difficulty": 3
    },
    {
      "id": 1721,
      "section": "Section 55 — Voice, Vision & Computer-Use Agents",
      "question": "What is a wake word system, and why is it a separate always-on model rather than continuous ASR?",
      "difficulty": 2
    },
    {
      "id": 1722,
      "section": "Section 55 — Voice, Vision & Computer-Use Agents",
      "question": "How does streaming ASR differ from batch ASR, and what does partial-hypothesis instability mean for a downstream LLM?",
      "difficulty": 3
    },
    {
      "id": 1723,
      "section": "Section 55 — Voice, Vision & Computer-Use Agents",
      "question": "Your voice agent works in the demo and fails in a call centre. Enumerate what changed.",
      "difficulty": 3
    },
    {
      "id": 1724,
      "section": "Section 55 — Voice, Vision & Computer-Use Agents",
      "question": "How do you handle a caller interrupting with a correction mid-sentence (\"no, the other one\") in a voice agent's state machine?",
      "difficulty": 3
    },
    {
      "id": 1725,
      "section": "Section 55 — Voice, Vision & Computer-Use Agents",
      "question": "What is speaker diarization, and when does a voice agent actually need it?",
      "difficulty": 2
    },
    {
      "id": 1726,
      "section": "Section 55 — Voice, Vision & Computer-Use Agents",
      "question": "Design evaluation for a voice agent. Why is transcript-level accuracy insufficient?",
      "difficulty": 3
    },
    {
      "id": 1727,
      "section": "Section 55 — Voice, Vision & Computer-Use Agents",
      "question": "Explain the three grounding strategies for computer-use agents — screenshot pixels, DOM, and accessibility tree. Compare them.",
      "difficulty": 3
    },
    {
      "id": 1728,
      "section": "Section 55 — Voice, Vision & Computer-Use Agents",
      "question": "Why is a hallucinated click categorically more dangerous than a hallucinated sentence, and what follows architecturally?",
      "difficulty": 3
    },
    {
      "id": 1729,
      "section": "Section 55 — Voice, Vision & Computer-Use Agents",
      "question": "Design action verification for a browser agent about to submit a purchase. What must be true before the click fires?",
      "difficulty": 3
    },
    {
      "id": 1730,
      "section": "Section 55 — Voice, Vision & Computer-Use Agents",
      "question": "A web page changes its layout. Explain why selector-based and vision-based agents fail differently, and which recovers better.",
      "difficulty": 3
    },
    {
      "id": 1731,
      "section": "Section 55 — Voice, Vision & Computer-Use Agents",
      "question": "How does prompt injection work against a computer-use agent, and why is it harder to defend than in a chat product?",
      "difficulty": 3
    },
    {
      "id": 1732,
      "section": "Section 55 — Voice, Vision & Computer-Use Agents",
      "question": "Explain the perceive-decide-act loop latency problem for GUI agents. Why can't you just screenshot every 100ms?",
      "difficulty": 3
    },
    {
      "id": 1733,
      "section": "Section 55 — Voice, Vision & Computer-Use Agents",
      "question": "Design the permission model for an agent with access to a logged-in browser session.",
      "difficulty": 3
    },
    {
      "id": 1734,
      "section": "Section 55 — Voice, Vision & Computer-Use Agents",
      "question": "What is a set-of-marks / element-labelling approach in vision-based GUI agents, and what problem does it solve?",
      "difficulty": 3
    },
    {
      "id": 1735,
      "section": "Section 55 — Voice, Vision & Computer-Use Agents",
      "question": "How would you evaluate a computer-use agent, given that task success is binary and rare?",
      "difficulty": 3
    },
    {
      "id": 1736,
      "section": "Section 55 — Voice, Vision & Computer-Use Agents",
      "question": "Your browser agent gets stuck in a loop clicking the same element. Diagnose systematically.",
      "difficulty": 3
    },
    {
      "id": 1737,
      "section": "Section 55 — Voice, Vision & Computer-Use Agents",
      "question": "Compare running a computer-use agent in a sandboxed VM versus the user's real browser session.",
      "difficulty": 3
    },
    {
      "id": 1738,
      "section": "Section 55 — Voice, Vision & Computer-Use Agents",
      "question": "What is a multimodal document parsing pipeline (Docling/GroundX-style), and why does naive PDF text extraction fail?",
      "difficulty": 3
    },
    {
      "id": 1739,
      "section": "Section 55 — Voice, Vision & Computer-Use Agents",
      "question": "Explain why table extraction is disproportionately hard, and how it breaks downstream RAG.",
      "difficulty": 3
    },
    {
      "id": 1740,
      "section": "Section 55 — Voice, Vision & Computer-Use Agents",
      "question": "How do you handle a document where the answer lives in a chart or diagram rather than text?",
      "difficulty": 3
    },
    {
      "id": 1741,
      "section": "Section 55 — Voice, Vision & Computer-Use Agents",
      "question": "Design a voice-driven RAG assistant end to end, and state which component you'd expect to fail first in production.",
      "difficulty": 3
    },
    {
      "id": 1742,
      "section": "Section 56 — Agent Memory & Context Engineering",
      "question": "Distinguish context, working memory, and long-term memory in an agent. Which is which in a real implementation?",
      "difficulty": 3
    },
    {
      "id": 1743,
      "section": "Section 56 — Agent Memory & Context Engineering",
      "question": "Compare episodic, semantic, and procedural memory for agents. Give a concrete implementation of each.",
      "difficulty": 3
    },
    {
      "id": 1744,
      "section": "Section 56 — Agent Memory & Context Engineering",
      "question": "What is context engineering, and how does it differ from prompt engineering?",
      "difficulty": 3
    },
    {
      "id": 1745,
      "section": "Section 56 — Agent Memory & Context Engineering",
      "question": "Your agent's context grows every turn until it hits the window limit. Enumerate your options in order.",
      "difficulty": 3
    },
    {
      "id": 1746,
      "section": "Section 56 — Agent Memory & Context Engineering",
      "question": "Explain rolling summarisation, and the specific information it systematically destroys.",
      "difficulty": 3
    },
    {
      "id": 1747,
      "section": "Section 56 — Agent Memory & Context Engineering",
      "question": "Design a memory system that decides what is worth remembering. What is the write policy?",
      "difficulty": 3
    },
    {
      "id": 1748,
      "section": "Section 56 — Agent Memory & Context Engineering",
      "question": "What is a temporal knowledge graph for agent memory (Graphiti-style), and what does it solve that vector memory doesn't?",
      "difficulty": 3
    },
    {
      "id": 1749,
      "section": "Section 56 — Agent Memory & Context Engineering",
      "question": "How do you handle contradictory memories — the user said X in March and not-X in June?",
      "difficulty": 3
    },
    {
      "id": 1750,
      "section": "Section 56 — Agent Memory & Context Engineering",
      "question": "Explain memory retrieval as a ranking problem. What signals beyond semantic similarity matter?",
      "difficulty": 3
    },
    {
      "id": 1751,
      "section": "Section 56 — Agent Memory & Context Engineering",
      "question": "What is context rot / lost-in-the-middle, and how does it change how you order retrieved content?",
      "difficulty": 3
    },
    {
      "id": 1752,
      "section": "Section 56 — Agent Memory & Context Engineering",
      "question": "Design memory for a multi-user agent where memories must never leak across users.",
      "difficulty": 3
    },
    {
      "id": 1753,
      "section": "Section 56 — Agent Memory & Context Engineering",
      "question": "When should a fact live in memory versus be re-derived from a source of truth?",
      "difficulty": 3
    },
    {
      "id": 1754,
      "section": "Section 56 — Agent Memory & Context Engineering",
      "question": "Explain the difference between an agent's scratchpad and its memory, and why conflating them causes bugs.",
      "difficulty": 3
    },
    {
      "id": 1755,
      "section": "Section 56 — Agent Memory & Context Engineering",
      "question": "How do you evaluate a memory system? What does \"good memory\" mean measurably?",
      "difficulty": 3
    },
    {
      "id": 1756,
      "section": "Section 56 — Agent Memory & Context Engineering",
      "question": "What is prompt caching / prefix caching, and how should it shape the way you order your context?",
      "difficulty": 3
    },
    {
      "id": 1757,
      "section": "Section 56 — Agent Memory & Context Engineering",
      "question": "Your agent remembers something wrong and keeps repeating it. Design the correction path.",
      "difficulty": 3
    },
    {
      "id": 1758,
      "section": "Section 56 — Agent Memory & Context Engineering",
      "question": "Explain the cost model of memory: what does a memory system actually cost per turn at scale?",
      "difficulty": 3
    },
    {
      "id": 1759,
      "section": "Section 56 — Agent Memory & Context Engineering",
      "question": "How do you decide memory retention and deletion policy under GDPR right-to-erasure?",
      "difficulty": 3
    },
    {
      "id": 1760,
      "section": "Section 56 — Agent Memory & Context Engineering",
      "question": "Compare storing raw conversation turns versus extracted facts as the memory substrate.",
      "difficulty": 3
    },
    {
      "id": 1761,
      "section": "Section 56 — Agent Memory & Context Engineering",
      "question": "Design the memory layer for a coding agent working across a long session in a large repository.",
      "difficulty": 3
    },
    {
      "id": 1762,
      "section": "Section 57 — Conformal Prediction & Uncertainty Quantification",
      "question": "What guarantee does conformal prediction actually provide, and what does it not?",
      "difficulty": 2
    },
    {
      "id": 1763,
      "section": "Section 57 — Conformal Prediction & Uncertainty Quantification",
      "question": "Explain split (inductive) conformal prediction step by step.",
      "difficulty": 1
    },
    {
      "id": 1764,
      "section": "Section 57 — Conformal Prediction & Uncertainty Quantification",
      "question": "What is a nonconformity score, and how does the choice of score affect the result?",
      "difficulty": 2
    },
    {
      "id": 1765,
      "section": "Section 57 — Conformal Prediction & Uncertainty Quantification",
      "question": "Distinguish marginal coverage from conditional coverage. Why does the difference matter in practice?",
      "difficulty": 3
    },
    {
      "id": 1766,
      "section": "Section 57 — Conformal Prediction & Uncertainty Quantification",
      "question": "What is exchangeability, and what happens to the guarantee when it is violated?",
      "difficulty": 3
    },
    {
      "id": 1767,
      "section": "Section 57 — Conformal Prediction & Uncertainty Quantification",
      "question": "Compare conformal prediction with Platt scaling and isotonic regression for calibration.",
      "difficulty": 2
    },
    {
      "id": 1768,
      "section": "Section 57 — Conformal Prediction & Uncertainty Quantification",
      "question": "How do you produce conformal prediction sets for a multi-class classifier, and how do you control set size?",
      "difficulty": 2
    },
    {
      "id": 1769,
      "section": "Section 57 — Conformal Prediction & Uncertainty Quantification",
      "question": "How does conformal prediction work for regression, and what makes the intervals adaptive?",
      "difficulty": 2
    },
    {
      "id": 1770,
      "section": "Section 57 — Conformal Prediction & Uncertainty Quantification",
      "question": "Explain Mondrian (class-conditional or group-conditional) conformal prediction and when it is required.",
      "difficulty": 3
    },
    {
      "id": 1771,
      "section": "Section 57 — Conformal Prediction & Uncertainty Quantification",
      "question": "How would you apply conformal prediction under distribution shift or to time-series data?",
      "difficulty": 3
    },
    {
      "id": 1772,
      "section": "Section 57 — Conformal Prediction & Uncertainty Quantification",
      "question": "Design a conformal abstention policy for a high-stakes classifier with a human reviewer.",
      "difficulty": 3
    },
    {
      "id": 1773,
      "section": "Section 57 — Conformal Prediction & Uncertainty Quantification",
      "question": "How do you size the calibration set, and what does that imply about achievable confidence levels?",
      "difficulty": 3
    },
    {
      "id": 1774,
      "section": "Section 57 — Conformal Prediction & Uncertainty Quantification",
      "question": "Can conformal prediction be applied to LLM outputs? Explain what is hard about it.",
      "difficulty": 3
    },
    {
      "id": 1775,
      "section": "Section 57 — Conformal Prediction & Uncertainty Quantification",
      "question": "Compare conformal prediction with Bayesian credible intervals and with deep ensembles.",
      "difficulty": 3
    },
    {
      "id": 1776,
      "section": "Section 57 — Conformal Prediction & Uncertainty Quantification",
      "question": "Distinguish aleatoric from epistemic uncertainty, and explain which methods address which.",
      "difficulty": 2
    },
    {
      "id": 1777,
      "section": "Section 57 — Conformal Prediction & Uncertainty Quantification",
      "question": "What is Monte Carlo dropout, and what are its limitations as an uncertainty estimate?",
      "difficulty": 1
    },
    {
      "id": 1778,
      "section": "Section 57 — Conformal Prediction & Uncertainty Quantification",
      "question": "How do you evaluate an uncertainty quantification method? What do you actually measure?",
      "difficulty": 2
    },
    {
      "id": 1779,
      "section": "Section 57 — Conformal Prediction & Uncertainty Quantification",
      "question": "A regulator asks you to guarantee your model's confidence claims. What do you offer and what do you refuse?",
      "difficulty": 3
    },
    {
      "id": 1780,
      "section": "Section 57 — Conformal Prediction & Uncertainty Quantification",
      "question": "How would you explain a conformal prediction set to a non-technical business stakeholder?",
      "difficulty": 2
    },
    {
      "id": 1781,
      "section": "Section 57 — Conformal Prediction & Uncertainty Quantification",
      "question": "Where does conformal prediction fail or mislead, and when would you not use it?",
      "difficulty": 3
    },
    {
      "id": 1782,
      "section": "Section 58 — Optimisation & Operations Research for AI Systems",
      "question": "How do you recognise that a problem is an optimisation problem rather than a prediction problem?",
      "difficulty": 2
    },
    {
      "id": 1783,
      "section": "Section 58 — Optimisation & Operations Research for AI Systems",
      "question": "Explain linear programming and what makes a problem linear.",
      "difficulty": 1
    },
    {
      "id": 1784,
      "section": "Section 58 — Optimisation & Operations Research for AI Systems",
      "question": "What changes when you add integer variables, and why does that matter for solve time?",
      "difficulty": 2
    },
    {
      "id": 1785,
      "section": "Section 58 — Optimisation & Operations Research for AI Systems",
      "question": "Explain how branch-and-bound solves a mixed-integer program.",
      "difficulty": 3
    },
    {
      "id": 1786,
      "section": "Section 58 — Optimisation & Operations Research for AI Systems",
      "question": "What is LP relaxation and how is it used in practice?",
      "difficulty": 2
    },
    {
      "id": 1787,
      "section": "Section 58 — Optimisation & Operations Research for AI Systems",
      "question": "Explain duality and what the shadow price tells you about a business constraint.",
      "difficulty": 3
    },
    {
      "id": 1788,
      "section": "Section 58 — Optimisation & Operations Research for AI Systems",
      "question": "Compare mixed-integer programming with constraint programming. When would you choose each?",
      "difficulty": 3
    },
    {
      "id": 1789,
      "section": "Section 58 — Optimisation & Operations Research for AI Systems",
      "question": "When would you use a metaheuristic instead of an exact solver?",
      "difficulty": 2
    },
    {
      "id": 1790,
      "section": "Section 58 — Optimisation & Operations Research for AI Systems",
      "question": "Design the predict-then-optimise architecture, and explain where it goes wrong.",
      "difficulty": 3
    },
    {
      "id": 1791,
      "section": "Section 58 — Optimisation & Operations Research for AI Systems",
      "question": "What is decision-focused learning, and when is it worth the complexity?",
      "difficulty": 3
    },
    {
      "id": 1792,
      "section": "Section 58 — Optimisation & Operations Research for AI Systems",
      "question": "How do you handle uncertainty in an optimisation model? Compare stochastic and robust optimisation.",
      "difficulty": 3
    },
    {
      "id": 1793,
      "section": "Section 58 — Optimisation & Operations Research for AI Systems",
      "question": "A stakeholder says the solver returned \"infeasible\". How do you diagnose and respond?",
      "difficulty": 2
    },
    {
      "id": 1794,
      "section": "Section 58 — Optimisation & Operations Research for AI Systems",
      "question": "How do you handle multiple competing objectives in an optimisation model?",
      "difficulty": 2
    },
    {
      "id": 1795,
      "section": "Section 58 — Optimisation & Operations Research for AI Systems",
      "question": "Design a workforce or shift scheduling system end to end.",
      "difficulty": 3
    },
    {
      "id": 1796,
      "section": "Section 58 — Optimisation & Operations Research for AI Systems",
      "question": "Design a vehicle routing system, and explain why VRP is hard.",
      "difficulty": 3
    },
    {
      "id": 1797,
      "section": "Section 58 — Optimisation & Operations Research for AI Systems",
      "question": "Explain the assignment problem and where it appears in AI systems.",
      "difficulty": 1
    },
    {
      "id": 1798,
      "section": "Section 58 — Optimisation & Operations Research for AI Systems",
      "question": "Compare commercial and open-source solvers. How would you make the buy decision?",
      "difficulty": 2
    },
    {
      "id": 1799,
      "section": "Section 58 — Optimisation & Operations Research for AI Systems",
      "question": "Where should an LLM sit in an optimisation workflow, and where must it not?",
      "difficulty": 3
    },
    {
      "id": 1800,
      "section": "Section 58 — Optimisation & Operations Research for AI Systems",
      "question": "How do you make an optimisation result explainable and trustworthy to the business?",
      "difficulty": 2
    },
    {
      "id": 1801,
      "section": "Section 58 — Optimisation & Operations Research for AI Systems",
      "question": "How do you deploy, monitor and maintain an optimisation model in production?",
      "difficulty": 2
    }
  ]
}
