[{"s":"3d-reconstruction","t":"3D Reconstruction","r":"vision","l":"3","d":"Creating 3D models from 2D images using geometry and deep learning.","n":5},{"s":"ab-testing","t":"A/B Testing","r":"evaluation","l":"2","d":"Comparing two model versions in production by routing traffic to each and measuring performance differences.","n":4},{"s":"accuracy","t":"Accuracy","r":"evaluation","l":"1","d":"The share of predictions a classifier gets right. Simple and intuitive, but misleading when one class is mu…","n":3},{"s":"activation-function","t":"Activation Function","r":"neural-nets","l":"1","d":"A non-linear function applied to neuron outputs that introduces non-linearity, enabling networks to learn c…","n":2},{"s":"active-learning","t":"Active Learning","r":"foundations","l":"3","d":"Iteratively selecting the most informative unlabeled examples for annotation to efficiently improve models…","n":4},{"s":"actor-critic","t":"Actor-Critic","r":"agents","l":"2","d":"RL architecture with two components: an actor (policy) that selects actions and a critic (value function) t…","n":11},{"s":"adagrad","t":"AdaGrad","r":"training","l":"3","d":"An optimizer that adapts learning rates for each parameter based on historical gradients, useful for sparse…","n":5},{"s":"adam","t":"Adam Optimizer","r":"training","l":"2","d":"An adaptive learning rate optimization algorithm combining momentum and RMSprop, widely used for training n…","n":7},{"s":"adamw","t":"AdamW","r":"training","l":"2","d":"Adam with decoupled weight decay, providing better regularization and often superior performance.","n":12},{"s":"adversarial-attack","t":"Adversarial Attack","r":"shipping","l":"2","d":"Intentionally crafted inputs designed to fool AI models into making incorrect predictions, exposing vulnera…","n":5},{"s":"adversarial-example","t":"Adversarial Example","r":"shipping","l":"3","d":"An input with imperceptible perturbations that causes a model to make a wrong prediction, highlighting mode…","n":6},{"s":"adversarial-perturbation","t":"Adversarial Perturbation","r":"shipping","l":"2","d":"Small carefully crafted changes to input that fool models while imperceptible to humans.","n":7},{"s":"adversarial-training","t":"Adversarial Training","r":"shipping","l":"2","d":"Training on adversarial examples to improve model robustness against attacks.","n":8},{"s":"agent","t":"Agent","r":"agents","l":"1","d":"In reinforcement learning, the learner and decision-maker that observes an environment, takes actions and a…","n":2},{"s":"ai-agent","t":"AI Agent","r":"agents","l":"1","d":"A system where a large language model decides its own next steps in a loop: calling tools, reading the resu…","n":20},{"s":"ai-alignment","t":"AI Alignment","r":"shipping","l":"2","d":"Ensuring AI systems behave in accordance with human values and intentions, a central challenge in AI safety.","n":1},{"s":"ai-ethics","t":"AI Ethics","r":"shipping","l":"2","d":"The study and practice of building and using AI in ways that respect people: fair, transparent, accountable…","n":0},{"s":"ai-governance","t":"AI Governance","r":"shipping","l":"2","d":"Policies, frameworks, and practices for responsible development and deployment of AI systems.","n":1},{"s":"ai-safety","t":"AI Safety","r":"shipping","l":"2","d":"Research and practices aimed at ensuring AI systems are safe, reliable, and beneficial, especially as capab…","n":0},{"s":"algorithmic-accountability","t":"Algorithmic Accountability","r":"shipping","l":"3","d":"Ensuring AI systems can be held accountable for their decisions and impacts.","n":2},{"s":"alignment-tax","t":"Alignment Tax","r":"language","l":"3","d":"Performance degradation that may occur when making models safer and more aligned with human values.","n":23},{"s":"alphafold","t":"AlphaFold","r":"foundations","l":"3","d":"DeepMind's breakthrough AI system for predicting protein structures with near-experimental accuracy.","n":4},{"s":"alphago","t":"AlphaGo","r":"agents","l":"3","d":"DeepMind's Go-playing AI that defeated world champions using deep RL and tree search.","n":4},{"s":"anaphora-resolution","t":"Anaphora Resolution","r":"language","l":"3","d":"Determining what a pronoun or noun phrase refers back to in text.","n":4},{"s":"anchor-box","t":"Anchor Box","r":"vision","l":"3","d":"Predefined boxes of various sizes and ratios serving as references for object detection.","n":6},{"s":"anomaly-detection","t":"Anomaly Detection","r":"foundations","l":"2","d":"Identifying unusual patterns or outliers in data that don't conform to expected behavior, used for fraud de…","n":2},{"s":"assistant-response","t":"Assistant Response","r":"language","l":"3","d":"The output generated by the language model in response to user prompts.","n":18},{"s":"attention-head","t":"Attention Head","r":"language","l":"2","d":"An individual attention mechanism in multi-head attention, learning specific patterns of relationships betw…","n":10},{"s":"attention-paper","t":"Attention Is All You Need","r":"language","l":"2","d":"The seminal 2017 paper by Vaswani et al. introducing the Transformer architecture that revolutionized NLP.","n":9},{"s":"attention-mask","t":"Attention Mask","r":"language","l":"3","d":"A binary mask indicating which tokens should be attended to, used to handle padding and causal masking.","n":14},{"s":"attention-mechanism","t":"Attention Mechanism","r":"language","l":"1","d":"A technique that lets a neural network weigh every part of its input when producing each output, focusing o…","n":7},{"s":"attention-score","t":"Attention Score","r":"language","l":"3","d":"The weight determining how much each value contributes to the output, computed from query-key similarity.","n":11},{"s":"attention-visualization","t":"Attention Visualization","r":"evaluation","l":"3","d":"Visualizing attention weights to understand which inputs the model focuses on.","n":9},{"s":"auc","t":"AUC","r":"evaluation","l":"2","d":"Area Under the Curve - measures the area under the ROC curve, indicating classification model performance (…","n":9},{"s":"audio-processing","t":"Audio Processing","r":"vision","l":"2","d":"Techniques for analyzing, transforming, and understanding audio signals for tasks like speech recognition a…","n":3},{"s":"autoencoder","t":"Autoencoder","r":"neural-nets","l":"2","d":"An unsupervised neural network that learns to compress data into a latent representation and reconstruct it…","n":3},{"s":"automl","t":"AutoML","r":"training","l":"3","d":"Automated Machine Learning - automating the process of model selection, architecture search, and hyperparam…","n":9},{"s":"autonomous-vehicles","t":"Autonomous Vehicles","r":"vision","l":"3","d":"Self-driving cars using computer vision, sensor fusion, and decision-making AI for navigation and control.","n":6},{"s":"autoregressive","t":"Autoregressive Model","r":"language","l":"2","d":"A model that generates output one token at a time, using previously generated tokens as input for the next…","n":4},{"s":"average-precision","t":"Average Precision","r":"evaluation","l":"2","d":"The weighted mean of precisions at each threshold, where the weight is the increase in recall from the prev…","n":7},{"s":"backdoor-attack","t":"Backdoor Attack","r":"shipping","l":"3","d":"Maliciously training models to behave normally except when specific triggers are present.","n":5},{"s":"backpropagation","t":"Backpropagation","r":"training","l":"1","d":"The algorithm for computing gradients of the loss with respect to network weights, enabling training throug…","n":6},{"s":"backward-pass","t":"Backward Pass","r":"neural-nets","l":"2","d":"The process of computing gradients by propagating error signals backward through the network during training.","n":7},{"s":"bagging","t":"Bagging","r":"foundations","l":"2","d":"Bootstrap aggregating: train many copies of a model on random resamples of the data, then average or vote t…","n":6},{"s":"bart","t":"BART","r":"language","l":"3","d":"Bidirectional and Auto-Regressive Transformer - combines BERT-like encoder with GPT-like decoder for sequen…","n":12},{"s":"baseline","t":"Baseline Model","r":"evaluation","l":"2","d":"A simple reference model (random, majority class, simple heuristic) used to benchmark more complex models a…","n":3},{"s":"batch-gd","t":"Batch Gradient Descent","r":"training","l":"3","d":"Computing gradients using the entire dataset, providing stable but slow updates.","n":4},{"s":"batch-normalization","t":"Batch Normalization","r":"neural-nets","l":"2","d":"A technique that normalizes layer inputs to stabilize and accelerate training by reducing internal covariat…","n":7},{"s":"batch-processing","t":"Batch Processing","r":"shipping","l":"3","d":"Processing multiple predictions together in batches rather than one at a time, improving throughput efficie…","n":3},{"s":"batch-size","t":"Batch Size","r":"training","l":"2","d":"The number of training examples processed together in one forward/backward pass.","n":4},{"s":"bayesian-inference","t":"Bayesian Inference","r":"foundations","l":"2","d":"Using Bayes' theorem to update beliefs about parameters given data, incorporating uncertainty.","n":0},{"s":"beam-search","t":"Beam Search","r":"language","l":"2","d":"A generation algorithm that maintains top-k candidates at each step, balancing quality and diversity.","n":11},{"s":"behavioral-cloning","t":"Behavioral Cloning","r":"agents","l":"3","d":"Supervised learning of a policy from state-action pairs in expert demonstrations.","n":4},{"s":"bellman-equation","t":"Bellman Equation","r":"agents","l":"2","d":"The recursive rule at the heart of reinforcement learning: a state's value equals the reward you get now pl…","n":8},{"s":"benchmark","t":"Benchmark","r":"evaluation","l":"2","d":"A standardized dataset and task used to compare model performance across different approaches (ImageNet, GL…","n":1},{"s":"benchmark-gaming","t":"Benchmark Gaming","r":"evaluation","l":"3","d":"Optimizing models specifically for benchmark performance rather than real-world capabilities, inflating sco…","n":2},{"s":"bert","t":"BERT","r":"language","l":"2","d":"Bidirectional Encoder Representations from Transformers - a model that understands context by looking at te…","n":21},{"s":"bias-parameter","t":"Bias","r":"neural-nets","l":"3","d":"A learnable offset added to neuron inputs, allowing the model to fit data that doesn't pass through the ori…","n":3},{"s":"bias-in-ai","t":"Bias in AI","r":"shipping","l":"2","d":"Systematic errors or unfair outcomes in AI systems, often reflecting biases in training data or model design.","n":2},{"s":"bias-variance-tradeoff","t":"Bias-Variance Tradeoff","r":"foundations","l":"2","d":"The balance between a model's bias (systematic error) and variance (sensitivity to training data fluctuatio…","n":4},{"s":"bidirectional-attention","t":"Bidirectional Attention","r":"language","l":"3","d":"Allowing tokens to attend to both past and future context, used in encoder models like BERT.","n":9},{"s":"black-box","t":"Black Box","r":"shipping","l":"2","d":"A model whose internal workings are difficult to understand or interpret, common with complex neural networks.","n":2},{"s":"bleu","t":"BLEU Score","r":"evaluation","l":"2","d":"A metric for evaluating machine translation quality by comparing n-gram overlap between generated and refer…","n":7},{"s":"blue-green-deployment","t":"Blue-Green Deployment","r":"shipping","l":"3","d":"Running two identical environments to enable zero-downtime model updates and rollbacks.","n":5},{"s":"boltzmann-machine","t":"Boltzmann Machine","r":"neural-nets","l":"3","d":"A stochastic recurrent neural network that can learn probability distributions over binary data.","n":6},{"s":"bos-token","t":"BOS Token","r":"language","l":"3","d":"Beginning Of Sequence token - marks the start of a sequence in language models.","n":3},{"s":"bottleneck","t":"Bottleneck","r":"neural-nets","l":"3","d":"A layer or section with reduced dimensions that compresses information, used in autoencoders and efficient…","n":7},{"s":"bounding-box","t":"Bounding Box","r":"vision","l":"2","d":"A rectangular box defined by coordinates that localizes an object in an image, used in object detection.","n":4},{"s":"bpe","t":"BPE","r":"language","l":"3","d":"Byte Pair Encoding - a subword tokenization algorithm that iteratively merges frequent character pairs to c…","n":3},{"s":"calibration","t":"Calibration","r":"evaluation","l":"2","d":"Ensuring predicted probabilities accurately reflect true likelihood of outcomes.","n":3},{"s":"canary-deployment","t":"Canary Deployment","r":"shipping","l":"3","d":"Gradually rolling out new model versions to subset of traffic before full deployment.","n":5},{"s":"capsnet","t":"Capsule Network","r":"neural-nets","l":"3","d":"An architecture using capsules (groups of neurons) that preserve spatial relationships, addressing limitati…","n":4},{"s":"catastrophic-forgetting","t":"Catastrophic Forgetting","r":"training","l":"2","d":"The tendency of neural networks to completely forget previously learned information when learning new tasks.","n":7},{"s":"causal-inference","t":"Causal Inference","r":"foundations","l":"3","d":"Determining cause-and-effect relationships from data, going beyond correlation to understand causal mechani…","n":0},{"s":"causal-language-modeling","t":"Causal Language Modeling","r":"language","l":"2","d":"Training a model to predict the next token given previous tokens, the foundation of autoregressive models l…","n":10},{"s":"causal-mask","t":"Causal Mask","r":"language","l":"3","d":"An attention mask ensuring tokens can only attend to previous positions, crucial for autoregressive generat…","n":18},{"s":"center-crop","t":"Center Crop","r":"vision","l":"3","d":"Extracting the central region of an image, often used during inference.","n":7},{"s":"certified-robustness","t":"Certified Robustness","r":"shipping","l":"3","d":"Provable guarantees that a model's prediction won't change within a specified input perturbation.","n":10},{"s":"chain-of-thought","t":"Chain-of-Thought","r":"language","l":"2","d":"A prompting technique where the model explains its reasoning step-by-step before giving a final answer, imp…","n":19},{"s":"chatbot","t":"Chatbot","r":"language","l":"2","d":"A conversational AI system that interacts with users through natural language, powered by NLP and LLMs.","n":18},{"s":"ci-cd-ml","t":"CI/CD for ML","r":"shipping","l":"3","d":"Continuous integration and deployment practices adapted for machine learning pipelines.","n":5},{"s":"classification","t":"Classification","r":"foundations","l":"1","d":"A supervised learning task where the model assigns each input to one of a fixed set of categories, such as…","n":2},{"s":"clip","t":"CLIP","r":"vision","l":"2","d":"Contrastive Language-Image Pre-training - a model jointly trained on images and text, enabling zero-shot im…","n":12},{"s":"cloze-task","t":"Cloze Task","r":"language","l":"3","d":"A task where words are removed from text and must be predicted, used for evaluation and pre-training.","n":4},{"s":"clustering","t":"Clustering","r":"foundations","l":"2","d":"An unsupervised learning technique that groups similar data points together based on their features or char…","n":2},{"s":"code-generation","t":"Code Generation","r":"language","l":"2","d":"AI systems that write code from natural language descriptions, powered by models like Codex, GitHub Copilot…","n":18},{"s":"cohens-kappa","t":"Cohen's Kappa","r":"evaluation","l":"3","d":"A metric measuring agreement between raters/models accounting for chance agreement.","n":4},{"s":"collaborative-filtering","t":"Collaborative Filtering","r":"foundations","l":"3","d":"Recommendation technique using patterns from multiple users to predict preferences, assuming similar users…","n":3},{"s":"color-jittering","t":"Color Jittering","r":"vision","l":"3","d":"Randomly adjusting brightness, contrast, saturation, and hue for image augmentation.","n":9},{"s":"compute","t":"Compute","r":"shipping","l":"2","d":"Informal term for computational resources (GPUs, TPUs, time) required for training or running AI models.","n":2},{"s":"computer-vision","t":"Computer Vision","r":"vision","l":"1","d":"The field of AI that gets computers to extract meaning from images and video: what is in them, where it is,…","n":3},{"s":"concentration-of-measure","t":"Concentration of Measure","r":"foundations","l":"3","d":"A phenomenon where random variables in high-dimensional spaces concentrate around their mean or median, wit…","n":3},{"s":"concept-drift","t":"Concept Drift","r":"shipping","l":"2","d":"When the relationship between inputs and the correct output changes over time, so a model's learned mapping…","n":5},{"s":"confusion-matrix","t":"Confusion Matrix","r":"evaluation","l":"2","d":"A table showing true positives, true negatives, false positives, and false negatives for classification eva…","n":3},{"s":"conjugate-prior","t":"Conjugate Prior","r":"foundations","l":"3","d":"A prior that when combined with a likelihood results in a posterior of the same family, simplifying Bayesia…","n":3},{"s":"constituency-parsing","t":"Constituency Parsing","r":"language","l":"3","d":"Analyzing sentence structure into nested constituents (noun phrases, verb phrases, etc.).","n":3},{"s":"constitutional-ai","t":"Constitutional AI","r":"language","l":"2","d":"Training AI systems using principles and rules rather than only human feedback, developed by Anthropic for…","n":23},{"s":"content-moderation","t":"Content Moderation","r":"shipping","l":"3","d":"Using AI to automatically detect and filter inappropriate, harmful, or policy-violating content.","n":5},{"s":"context-window","t":"Context Window","r":"language","l":"1","d":"The maximum number of tokens a language model can take into account at once, counting both the input it rea…","n":10},{"s":"continual-learning","t":"Continual Learning","r":"training","l":"3","d":"Learning new tasks sequentially without forgetting previously learned tasks, addressing catastrophic forget…","n":8},{"s":"contrastive-learning","t":"Contrastive Learning","r":"training","l":"2","d":"A self-supervised learning approach that learns representations by contrasting similar and dissimilar examp…","n":9},{"s":"contrastive-loss","t":"Contrastive Loss","r":"training","l":"2","d":"A loss function for learning similarity metrics, bringing similar pairs together and separating dissimilar…","n":9},{"s":"conversation-history","t":"Conversation History","r":"language","l":"3","d":"Previous messages in a multi-turn dialogue, provided as context for coherent conversations.","n":11},{"s":"convolution","t":"Convolution","r":"neural-nets","l":"2","d":"A mathematical operation that applies filters/kernels to input data to extract features like edges, texture…","n":2},{"s":"cnn","t":"Convolutional Neural Network","r":"neural-nets","l":"1","d":"A neural network that scans images with small learned filters, reusing the same weights at every position t…","n":3},{"s":"coreference-resolution","t":"Coreference Resolution","r":"language","l":"3","d":"Identifying all expressions in text that refer to the same entity (e.g., linking pronouns to nouns).","n":3},{"s":"cosine-annealing","t":"Cosine Annealing","r":"training","l":"3","d":"A learning rate schedule following a cosine curve, smoothly decreasing the rate over training.","n":6},{"s":"cosine-similarity","t":"Cosine Similarity","r":"language","l":"2","d":"How closely two vectors point in the same direction, from 1 (same direction) to -1 (opposite): the standard…","n":7},{"s":"cov","t":"Covariance","r":"foundations","l":"3","d":"A measure of how two variables change together, indicating the direction of their linear relationship.","n":0},{"s":"cross-attention","t":"Cross-Attention","r":"language","l":"2","d":"Attention between two different sequences, where queries come from one and keys/values from another.","n":9},{"s":"cross-entropy","t":"Cross-Entropy Loss","r":"training","l":"2","d":"A loss function for classification that measures the difference between predicted and true probability dist…","n":7},{"s":"cross-validation","t":"Cross-Validation","r":"evaluation","l":"2","d":"A technique for assessing model performance by partitioning data into subsets, training on some and validat…","n":2},{"s":"ctc-loss","t":"CTC Loss","r":"training","l":"3","d":"Connectionist Temporal Classification loss for sequence tasks without alignment, used in speech recognition.","n":6},{"s":"curriculum-learning","t":"Curriculum Learning","r":"training","l":"3","d":"Training strategy where examples are presented from easy to hard, mimicking human learning for improved con…","n":2},{"s":"curse-of-dimensionality","t":"Curse of Dimensionality","r":"foundations","l":"2","d":"Phenomena where algorithms become inefficient as dimensionality increases, including data sparsity and dist…","n":2},{"s":"cutmix","t":"CutMix","r":"training","l":"3","d":"Data augmentation combining image patches and labels from two examples, improving robustness.","n":4},{"s":"cutout","t":"Cutout","r":"training","l":"3","d":"Data augmentation randomly masking out square regions of images during training.","n":4},{"s":"cyclical-lr","t":"Cyclical Learning Rate","r":"training","l":"3","d":"Varying learning rate between bounds in cycles, potentially escaping local minima.","n":6},{"s":"data-annotation","t":"Data Annotation","r":"foundations","l":"2","d":"Adding labels to raw data, such as tags, boxes, transcripts or rankings, usually by people, so models can l…","n":2},{"s":"data-augmentation","t":"Data Augmentation","r":"training","l":"2","d":"Creating variations of training data through transformations (rotation, cropping, noise) to improve model g…","n":3},{"s":"data-cleaning","t":"Data Cleaning","r":"foundations","l":"2","d":"Finding and fixing errors, inconsistencies and gaps in a dataset (duplicates, typos, missing values, wrong…","n":1},{"s":"data-drift","t":"Data Drift","r":"shipping","l":"2","d":"Changes in input data distribution over time that can degrade model performance in production.","n":3},{"s":"data-leakage","t":"Data Leakage","r":"evaluation","l":"2","d":"When information from outside the training data is used to create the model, leading to overly optimistic p…","n":2},{"s":"data-parallelism","t":"Data Parallelism","r":"training","l":"2","d":"Replicating the model across devices, each processing different data batches.","n":4},{"s":"data-pipeline","t":"Data Pipeline","r":"shipping","l":"2","d":"The automated sequence of steps that moves raw data from its sources, through checks and transformations, i…","n":2},{"s":"data-poisoning","t":"Data Poisoning","r":"shipping","l":"2","d":"Corrupting training data to manipulate model behavior or introduce vulnerabilities.","n":4},{"s":"data-preprocessing","t":"Data Preprocessing","r":"foundations","l":"2","d":"Cleaning, transforming, and preparing raw data for model training (handling missing values, normalization,…","n":1},{"s":"data-versioning","t":"Data Versioning","r":"shipping","l":"2","d":"Tracking different versions of datasets to ensure reproducibility and manage changes.","n":5},{"s":"dataset","t":"Dataset","r":"foundations","l":"2","d":"A collection of data examples used for training, validating, or testing machine learning models.","n":0},{"s":"decision-tree","t":"Decision Tree","r":"foundations","l":"2","d":"A tree-structured model that makes decisions by splitting data based on feature values, interpretable but p…","n":4},{"s":"decoder-only","t":"Decoder-Only Model","r":"language","l":"2","d":"A transformer architecture with only decoder layers, using causal masking for autoregressive generation (GP…","n":14},{"s":"decoding-strategies","t":"Decoding Strategies","r":"language","l":"2","d":"The rules a language model uses to turn its next-token probabilities into actual text, from always taking t…","n":9},{"s":"dbn","t":"Deep Belief Network","r":"neural-nets","l":"3","d":"A generative model composed of multiple layers of RBMs, historically important for unsupervised pre-training.","n":8},{"s":"deep-learning","t":"Deep Learning","r":"neural-nets","l":"1","d":"A subset of machine learning that uses neural networks with multiple layers (deep neural networks) to learn…","n":2},{"s":"dqn","t":"Deep Q-Network","r":"agents","l":"2","d":"Combining Q-learning with deep neural networks to handle high-dimensional state spaces, enabling RL for com…","n":8},{"s":"dependency-parsing","t":"Dependency Parsing","r":"language","l":"3","d":"Analyzing grammatical structure by identifying relationships between words (subject, object, modifier).","n":3},{"s":"depth-estimation","t":"Depth Estimation","r":"vision","l":"3","d":"Predicting distance of objects from the camera using monocular or stereo images.","n":4},{"s":"depthwise-separable-conv","t":"Depthwise Separable Convolution","r":"neural-nets","l":"3","d":"An efficient convolution that factorizes standard convolution into depthwise and pointwise steps, reducing…","n":3},{"s":"dialogue-state-tracking","t":"Dialogue State Tracking","r":"language","l":"3","d":"Maintaining a representation of conversation state in dialogue systems.","n":6},{"s":"dice-coefficient","t":"Dice Coefficient","r":"evaluation","l":"3","d":"A metric measuring overlap between predicted and ground truth segmentations, common in medical imaging.","n":11},{"s":"differential-privacy","t":"Differential Privacy","r":"shipping","l":"3","d":"A mathematical framework for quantifying and limiting privacy loss when releasing information about datasets.","n":3},{"s":"diffusion-model","t":"Diffusion Model","r":"vision","l":"1","d":"A generative model trained to remove noise a step at a time, so it can turn pure random noise into a new im…","n":2},{"s":"dilated-convolution","t":"Dilated Convolution","r":"neural-nets","l":"3","d":"Convolution with gaps between kernel elements, expanding the receptive field without increasing parameters.","n":4},{"s":"dimensionality-reduction","t":"Dimensionality Reduction","r":"foundations","l":"2","d":"Techniques to reduce the number of input features while preserving important information (PCA, t-SNE, autoe…","n":4},{"s":"discriminator","t":"Discriminator","r":"vision","l":"3","d":"In GANs, the network that tries to distinguish between real and generated data, providing training signal t…","n":6},{"s":"distillation-temperature","t":"Distillation Temperature","r":"training","l":"3","d":"A hyperparameter in knowledge distillation controlling how soft the teacher's outputs are.","n":13},{"s":"distributed-training","t":"Distributed Training","r":"training","l":"2","d":"Training one model across many GPUs or machines at once, by splitting the data, the model, or both, and kee…","n":3},{"s":"distribution-shift","t":"Distribution Shift","r":"evaluation","l":"2","d":"When the data a model meets in use differs from the data it was trained on, so accuracy measured before dep…","n":4},{"s":"domain-randomization","t":"Domain Randomization","r":"agents","l":"3","d":"Training with randomized simulation parameters to improve transfer to real-world environments.","n":9},{"s":"dropout","t":"Dropout","r":"neural-nets","l":"1","d":"A regularization technique that randomly switches off units during training so the network can't lean on an…","n":5},{"s":"drug-discovery","t":"Drug Discovery","r":"foundations","l":"3","d":"Using AI to accelerate pharmaceutical research by predicting molecular properties, protein structures, and…","n":3},{"s":"early-stopping","t":"Early Stopping","r":"training","l":"2","d":"Stopping training when validation performance stops improving, preventing overfitting.","n":5},{"s":"esn","t":"Echo State Network","r":"neural-nets","l":"3","d":"A recurrent network with a fixed random reservoir and trained readout layer, efficient for time series proc…","n":4},{"s":"edge-deployment","t":"Edge Deployment","r":"shipping","l":"2","d":"Running models on edge devices (phones, IoT) rather than cloud servers for lower latency and privacy.","n":4},{"s":"efficientnet","t":"EfficientNet","r":"vision","l":"3","d":"A family of CNNs that scale depth, width, and resolution simultaneously using compound scaling for optimal…","n":4},{"s":"elu","t":"ELU","r":"neural-nets","l":"3","d":"Exponential Linear Unit - an activation function that allows negative values, helping with vanishing gradie…","n":4},{"s":"embedding","t":"Embedding","r":"language","l":"1","d":"A list of numbers (a vector) that represents a word, sentence, image or other item, learned so that similar…","n":6},{"s":"emergent-abilities","t":"Emergent Abilities","r":"language","l":"2","d":"Capabilities that appear suddenly in large language models at certain scales, not present in smaller models.","n":20},{"s":"erm","t":"Empirical Risk Minimization","r":"training","l":"3","d":"The principle of choosing a model that minimizes error on training data, fundamental to supervised learning.","n":4},{"s":"encoder-decoder","t":"Encoder-Decoder","r":"language","l":"2","d":"A architecture where the encoder processes input and the decoder generates output, used in translation and…","n":11},{"s":"encoder-only","t":"Encoder-Only Model","r":"language","l":"3","d":"A transformer with only encoder layers and bidirectional attention, suited for understanding tasks (BERT fa…","n":11},{"s":"energy-based-model","t":"Energy-Based Model","r":"neural-nets","l":"2","d":"A model that scores every possible configuration with a single number, its energy, where low energy means c…","n":5},{"s":"ensemble-learning","t":"Ensemble Learning","r":"foundations","l":"2","d":"Combining multiple models to produce better predictions than any individual model (bagging, boosting, stack…","n":2},{"s":"entity-linking","t":"Entity Linking","r":"language","l":"3","d":"Linking entity mentions in text to entries in a knowledge base.","n":3},{"s":"entropy","t":"Entropy","r":"foundations","l":"2","d":"A measure of uncertainty or randomness in a random variable from information theory.","n":0},{"s":"environment","t":"Environment","r":"agents","l":"2","d":"In RL, the world the agent interacts with, providing states, accepting actions, and returning rewards.","n":3},{"s":"eos-token","t":"EOS Token","a":["Stop Token"],"r":"language","l":"3","d":"End Of Sequence token - signals when the model has finished generating a complete output.","n":3},{"s":"epoch","t":"Epoch","r":"training","l":"2","d":"One complete pass through the entire training dataset during the training process.","n":2},{"s":"elbo","t":"Evidence Lower Bound","r":"foundations","l":"3","d":"A lower bound on log likelihood used in variational inference and VAEs for tractable optimization.","n":4},{"s":"em-algorithm","t":"Expectation-Maximization","r":"foundations","l":"3","d":"An iterative algorithm for finding maximum likelihood estimates in models with latent variables.","n":2},{"s":"experiment-tracking","t":"Experiment Tracking","r":"shipping","l":"2","d":"Recording hyperparameters, metrics, and artifacts from training runs for comparison and reproducibility.","n":6},{"s":"explainability","t":"Explainability","a":["Explainable AI","XAI"],"r":"shipping","l":"2","d":"The ability to explain how an AI model makes decisions in human-understandable terms, crucial for trust and…","n":3},{"s":"exploding-gradient","t":"Exploding Gradient","r":"training","l":"3","d":"A problem where gradients become extremely large during backpropagation, causing unstable training and NaN…","n":7},{"s":"exploration-exploitation","t":"Exploration vs Exploitation","r":"agents","l":"2","d":"The RL dilemma of trying new actions (exploration) versus using known good actions (exploitation) to maximi…","n":2},{"s":"f1-score","t":"F1 Score","r":"evaluation","l":"2","d":"The harmonic mean of precision and recall, providing a single metric that balances both concerns.","n":6},{"s":"face-detection","t":"Face Detection","r":"vision","l":"3","d":"Locating faces in images, a precursor to recognition and analysis.","n":6},{"s":"facial-recognition","t":"Facial Recognition","r":"vision","l":"3","d":"Identifying or verifying people from face images using deep learning.","n":11},{"s":"fairness","t":"Fairness","r":"shipping","l":"2","d":"Ensuring AI systems treat all individuals and groups equitably, without discrimination based on protected a…","n":3},{"s":"false-negative","t":"False Negative","r":"evaluation","l":"2","d":"Positive cases that are incorrectly predicted as negative (Type II error) in classification.","n":4},{"s":"false-positive","t":"False Positive","r":"evaluation","l":"1","d":"A case the model labels positive that is actually negative, such as a legitimate email sent to spam. Statis…","n":4},{"s":"false-positive-rate","t":"False Positive Rate","r":"evaluation","l":"2","d":"The proportion of negatives incorrectly classified as positive.","n":6},{"s":"faster-r-cnn","t":"Faster R-CNN","r":"vision","l":"3","d":"An object detection architecture with RPN for efficient region proposals.","n":11},{"s":"feature","t":"Feature","r":"foundations","l":"1","d":"A single measurable property of an example, such as a house's floor area or how many links an email contain…","n":1},{"s":"feature-engineering","t":"Feature Engineering","r":"foundations","l":"2","d":"The process of selecting, transforming, and creating input features to improve model performance.","n":2},{"s":"feature-importance","t":"Feature Importance","r":"foundations","l":"3","d":"Measures indicating which features contribute most to model predictions, useful for interpretation and sele…","n":2},{"s":"feature-map","t":"Feature Map","r":"neural-nets","l":"2","d":"The output of applying a convolutional filter to an input, representing detected features at various spatia…","n":3},{"s":"fpn","t":"Feature Pyramid Network","r":"vision","l":"3","d":"A CNN architecture creating multi-scale feature representations for detecting objects at different sizes.","n":8},{"s":"feature-selection","t":"Feature Selection","r":"foundations","l":"2","d":"Choosing the most relevant features from available data to reduce dimensionality and improve model performa…","n":2},{"s":"feature-store","t":"Feature Store","r":"shipping","l":"3","d":"A centralized platform for managing, storing, and serving features for ML models.","n":7},{"s":"federated-learning","t":"Federated Learning","r":"shipping","l":"2","d":"Training models across decentralized devices holding local data, without exchanging the data itself, preser…","n":4},{"s":"feedforward-network","t":"Feedforward Network","r":"neural-nets","l":"2","d":"A neural network where information flows in one direction from input to output without cycles.","n":2},{"s":"few-shot","t":"Few-Shot Learning","r":"language","l":"2","d":"Learning to perform a task from a small number of examples provided in the prompt, without parameter updates.","n":19},{"s":"filter","t":"Filter","r":"neural-nets","l":"3","d":"Synonym for kernel - the learnable weight matrix applied in convolutions to extract features.","n":3},{"s":"fine-tuning","t":"Fine-Tuning","r":"training","l":"1","d":"The process of further training a pre-trained model on a specific dataset to adapt it for a particular task…","n":6},{"s":"flash-attention","t":"Flash Attention","r":"language","l":"3","d":"An efficient attention algorithm reducing memory usage and increasing speed through clever recomputation.","n":9},{"s":"flops","t":"FLOPS","r":"shipping","l":"3","d":"Floating Point Operations Per Second - a measure of computational performance, used to quantify training an…","n":3},{"s":"focal-loss","t":"Focal Loss","r":"training","l":"2","d":"A modified cross-entropy loss that down-weights easy examples, helping with class imbalance.","n":11},{"s":"forward-pass","t":"Forward Pass","r":"neural-nets","l":"2","d":"The process of passing input through the network to generate predictions during training or inference.","n":2},{"s":"foundation-model","t":"Foundation Model","r":"language","l":"1","d":"A large model pre-trained on broad data at scale that can be adapted to many downstream tasks, such as GPT-…","n":7},{"s":"fraud-detection","t":"Fraud Detection","r":"foundations","l":"3","d":"Identifying fraudulent transactions or activities using anomaly detection and pattern recognition in financ…","n":5},{"s":"function-calling","t":"Function Calling","r":"agents","l":"2","d":"The API mechanism behind LLM tool use: you describe functions with a schema, the model replies with a struc…","n":19},{"s":"gru","t":"Gated Recurrent Unit","r":"neural-nets","l":"2","d":"A simplified variant of LSTM with fewer parameters, using an update gate and a reset gate to control inform…","n":3},{"s":"gaussian-process","t":"Gaussian Process","r":"foundations","l":"3","d":"A non-parametric Bayesian approach for regression and classification, defining distributions over functions.","n":4},{"s":"gelu","t":"GELU","r":"neural-nets","l":"2","d":"Gaussian Error Linear Unit - a smooth activation function combining properties of dropout and ReLU, used in…","n":4},{"s":"generalization","t":"Generalization","r":"foundations","l":"1","d":"A model's ability to perform well on new data it never saw during training, which is the whole point of lea…","n":3},{"s":"gan","t":"Generative Adversarial Network","r":"vision","l":"2","d":"A framework where two networks (generator and discriminator) compete, with the generator learning to create…","n":5},{"s":"generator","t":"Generator","r":"vision","l":"3","d":"In GANs, the network that creates synthetic data attempting to fool the discriminator into thinking it's real.","n":6},{"s":"gibbs-sampling","t":"Gibbs Sampling","r":"foundations","l":"3","d":"An MCMC method that samples from conditional distributions to approximate joint distributions.","n":2},{"s":"gpt","t":"GPT","r":"language","l":"2","d":"Generative Pre-trained Transformer - an autoregressive language model architecture that predicts the next t…","n":19},{"s":"gpu","t":"GPU","r":"shipping","l":"2","d":"Graphics Processing Unit - hardware accelerator with thousands of cores, essential for parallel computation…","n":0},{"s":"grad-cam","t":"Grad-CAM","r":"evaluation","l":"3","d":"Gradient-weighted Class Activation Mapping - visualizing which image regions influenced CNN predictions.","n":11},{"s":"gradient-accumulation","t":"Gradient Accumulation","r":"training","l":"3","d":"Summing gradients over multiple batches before updating, simulating larger effective batch sizes.","n":8},{"s":"gradient-boosting","t":"Gradient Boosting","r":"foundations","l":"2","d":"An ensemble technique that builds models sequentially, each correcting errors of previous ones (XGBoost, Li…","n":6},{"s":"gradient-checkpointing","t":"Gradient Checkpointing","r":"training","l":"3","d":"Trading computation for memory by recomputing activations during backprop instead of storing them.","n":7},{"s":"gradient-clipping","t":"Gradient Clipping","r":"training","l":"3","d":"Limiting gradient magnitudes during training to prevent exploding gradients and stabilize training.","n":8},{"s":"gradient-descent","t":"Gradient Descent","r":"training","l":"1","d":"An optimization method that repeatedly moves a model's parameters a small step in the direction that most r…","n":3},{"s":"gat","t":"Graph Attention Network","r":"neural-nets","l":"3","d":"A GNN using attention mechanisms to weight neighbor contributions when aggregating information.","n":9},{"s":"gnn","t":"Graph Neural Network","r":"neural-nets","l":"2","d":"Neural networks designed to operate on graph-structured data, learning representations of nodes, edges, and…","n":2},{"s":"greedy-decoding","t":"Greedy Decoding","r":"language","l":"2","d":"Always selecting the most likely next token during generation, fast but can lead to repetitive or suboptima…","n":10},{"s":"ground-truth","t":"Ground Truth","r":"foundations","l":"3","d":"The correct or true labels/values for data, used as targets during training and evaluation benchmarks.","n":1},{"s":"grounding","t":"Grounding","r":"language","l":"2","d":"Connecting model outputs to real-world facts, sources, or evidence to improve factuality and reduce halluci…","n":19},{"s":"group-normalization","t":"Group Normalization","r":"neural-nets","l":"3","d":"Normalizing groups of channels independently, more stable than batch normalization for small batch sizes.","n":8},{"s":"grpc","t":"gRPC","r":"shipping","l":"3","d":"A high-performance RPC framework often used for low-latency model serving.","n":4},{"s":"hallucination","t":"Hallucination","r":"language","l":"1","d":"When a language model produces fluent, confident output that is false or unsupported by its sources, such a…","n":18},{"s":"he-initialization","t":"He Initialization","r":"neural-nets","l":"3","d":"Weight initialization designed for ReLU activations, preventing vanishing/exploding gradients in deep netwo…","n":6},{"s":"hidden-layer","t":"Hidden Layer","r":"neural-nets","l":"2","d":"Intermediate layers between input and output that learn hierarchical representations in neural networks.","n":3},{"s":"hmm","t":"Hidden Markov Model","r":"foundations","l":"3","d":"A statistical model with hidden states that transition probabilistically, generating observable outputs.","n":1},{"s":"hierarchical-rl","t":"Hierarchical RL","r":"agents","l":"3","d":"Learning policies at multiple levels of abstraction, with high-level goals and low-level skills.","n":4},{"s":"hinge-loss","t":"Hinge Loss","r":"training","l":"2","d":"A loss function for maximum-margin classification, used in SVMs.","n":7},{"s":"homomorphic-encryption","t":"Homomorphic Encryption","r":"shipping","l":"3","d":"Encryption allowing computation on encrypted data, enabling private model inference.","n":3},{"s":"hopfield-network","t":"Hopfield Network","r":"neural-nets","l":"3","d":"A recurrent network that serves as content-addressable memory, capable of pattern completion and associativ…","n":6},{"s":"huber-loss","t":"Huber Loss","r":"training","l":"2","d":"A loss function that's quadratic for small errors and linear for large errors, robust to outliers.","n":8},{"s":"hyperparameter","t":"Hyperparameter","r":"training","l":"1","d":"A setting chosen before training, such as the learning rate or batch size, that controls how a model learns…","n":5},{"s":"hyperparameter-tuning","t":"Hyperparameter Tuning","r":"training","l":"2","d":"The process of finding optimal hyperparameter values through techniques like grid search, random search, or…","n":8},{"s":"image-augmentation","t":"Image Augmentation","a":["Data Augmentation in Vision"],"r":"vision","l":"2","d":"Applying transformations (rotation, flip, crop, color) to increase training data diversity.","n":8},{"s":"image-classification","t":"Image Classification","r":"vision","l":"1","d":"Assigning one label from a fixed set of categories to a whole image, such as \"cat\" or \"defective part\"; a c…","n":6},{"s":"image-generation","t":"Image Generation","r":"vision","l":"1","d":"Creating new images, often from a text prompt, with generative models such as GANs, VAEs and diffusion mode…","n":4},{"s":"image-inpainting","t":"Image Inpainting","r":"vision","l":"3","d":"Filling in missing or corrupted regions of images using context and generative models.","n":5},{"s":"image-normalization","t":"Image Normalization","r":"vision","l":"3","d":"Scaling pixel values to standard ranges (e.g., mean=0, std=1) to improve training.","n":9},{"s":"image-preprocessing","t":"Image Preprocessing","r":"vision","l":"2","d":"Transforming images before model input (resizing, normalization, color adjustment).","n":6},{"s":"imagenet","t":"ImageNet","r":"vision","l":"2","d":"A large-scale dataset of 14M images in 20K categories, historically used as the benchmark for image classif…","n":8},{"s":"imbalanced-dataset","t":"Imbalanced Dataset","r":"foundations","l":"3","d":"A dataset where classes have significantly different numbers of examples, causing models to bias toward maj…","n":4},{"s":"imitation-learning","t":"Imitation Learning","r":"agents","l":"2","d":"Learning policies by mimicking expert behavior from demonstrations.","n":3},{"s":"in-context-learning","t":"In-Context Learning","r":"language","l":"2","d":"The ability of LLMs to learn from examples and instructions provided in the input prompt without training.","n":18},{"s":"inception","t":"Inception","r":"vision","l":"3","d":"A CNN architecture (GoogLeNet) using parallel convolutions of different sizes to capture multi-scale featur…","n":4},{"s":"inductive-bias","t":"Inductive Bias","r":"foundations","l":"3","d":"Assumptions built into a learning algorithm that guide it toward certain solutions over others.","n":2},{"s":"inference","t":"Inference","r":"shipping","l":"1","d":"Running a trained model on new inputs to get predictions, with its weights frozen: the stage of a model's l…","n":2},{"s":"inference-latency","t":"Inference Latency","r":"shipping","l":"2","d":"The time delay between submitting input and receiving output from a deployed model, critical for real-time…","n":3},{"s":"information-bottleneck","t":"Information Bottleneck","r":"foundations","l":"3","d":"A principle for learning representations that compress input while retaining information relevant to predic…","n":8},{"s":"information-extraction","t":"Information Extraction","r":"language","l":"2","d":"Automatically extracting structured information from unstructured text.","n":1},{"s":"information-retrieval","t":"Information Retrieval","r":"language","l":"2","d":"The field of finding the documents in a large collection that satisfy a person's information need, and rank…","n":1},{"s":"information-theory","t":"Information Theory","r":"foundations","l":"2","d":"The mathematics of measuring, compressing and transmitting information, founded by Claude Shannon in 1948 a…","n":0},{"s":"instance-segmentation","t":"Instance Segmentation","r":"vision","l":"2","d":"Combining object detection and segmentation to identify individual object instances at the pixel level.","n":9},{"s":"instruction-following","t":"Instruction Following","r":"language","l":"3","d":"The ability of language models to understand and execute instructions provided in prompts.","n":18},{"s":"instruction-tuning","t":"Instruction Tuning","r":"language","l":"2","d":"Fine-tuning LLMs on diverse instruction-following tasks to improve zero-shot performance on new instructions.","n":19},{"s":"intent-recognition","t":"Intent Recognition","r":"language","l":"3","d":"Identifying the user's intention or goal from their utterance in dialogue systems.","n":5},{"s":"interpretability","t":"Interpretability","r":"shipping","l":"2","d":"Understanding the internal workings of AI models, including which features influence predictions and why.","n":2},{"s":"iou","t":"Intersection over Union","a":["Jaccard Index"],"r":"evaluation","l":"1","d":"A score from 0 to 1 for how well a predicted region matches the true one: the area they share divided by th…","n":5},{"s":"inverse-rl","t":"Inverse Reinforcement Learning","r":"agents","l":"3","d":"Learning reward functions from expert demonstrations, inferring what is being optimized.","n":3},{"s":"js-divergence","t":"Jensen-Shannon Divergence","r":"foundations","l":"3","d":"A symmetric measure of similarity between probability distributions, related to KL divergence.","n":2},{"s":"kernel","t":"Kernel","r":"neural-nets","l":"2","d":"A small matrix of weights used in convolutional layers to detect specific features or patterns in input data.","n":3},{"s":"kernel-trick","t":"Kernel Trick","r":"foundations","l":"2","d":"Computing dot products in a huge feature space directly from the original inputs, so linear methods can lea…","n":4},{"s":"keypoint-detection","t":"Keypoint Detection","r":"vision","l":"2","d":"Finding specific, repeatable points in an image: distinctive spots for matching two views, or named landmar…","n":6},{"s":"kl-divergence","t":"KL Divergence","r":"foundations","l":"2","d":"Kullback-Leibler divergence - a measure of how one probability distribution differs from another.","n":1},{"s":"knowledge-distillation","t":"Knowledge Distillation","r":"training","l":"2","d":"Training a smaller 'student' model to mimic a larger 'teacher' model, transferring knowledge while reducing…","n":6},{"s":"knowledge-graph","t":"Knowledge Graph","r":"foundations","l":"3","d":"A structured representation of knowledge as entities and their relationships, used for reasoning and inform…","n":0},{"s":"l1-regularization","t":"L1 Regularization","r":"training","l":"3","d":"Adding the sum of absolute weights to the loss function, promoting sparsity and feature selection.","n":4},{"s":"l2-regularization","t":"L2 Regularization","r":"training","l":"2","d":"Adding the sum of squared weights to the loss function, penalizing large weights and improving generalization.","n":4},{"s":"label-smoothing","t":"Label Smoothing","r":"training","l":"3","d":"Softening target labels to prevent overconfidence and improve generalization.","n":8},{"s":"labeled-data","t":"Labeled Data","r":"foundations","l":"2","d":"Data with associated target outputs or annotations, required for supervised learning tasks.","n":1},{"s":"lamb","t":"LAMB Optimizer","r":"training","l":"3","d":"Layer-wise Adaptive Moments optimizer for Batch training - enables very large batch training for transformers.","n":9},{"s":"language-modeling","t":"Language Modeling","r":"language","l":"1","d":"Assigning probabilities to sequences of text, in practice by predicting each next token from the ones befor…","n":3},{"s":"language-understanding","t":"Language Understanding","r":"language","l":"2","d":"The ability to comprehend meaning, context, intent, and nuance in natural language.","n":1},{"s":"llm","t":"Large Language Model","r":"language","l":"1","d":"A neural network, almost always a transformer, trained on vast amounts of text to predict the next token, w…","n":17},{"s":"latent-diffusion","t":"Latent Diffusion","r":"vision","l":"2","d":"A diffusion model that runs its denoising in the compressed latent space of a pretrained autoencoder instea…","n":11},{"s":"lda","t":"Latent Dirichlet Allocation","r":"language","l":"3","d":"A generative probabilistic model for topic modeling that represents documents as mixtures of topics.","n":6},{"s":"latent-space","t":"Latent Space","r":"neural-nets","l":"2","d":"A compressed, learned representation space where similar data points are close together, used in autoencode…","n":7},{"s":"latent-variable","t":"Latent Variable","r":"foundations","l":"3","d":"Hidden or unobserved variables in a model that influence observed data but aren't directly measured.","n":0},{"s":"layer","t":"Layer","r":"neural-nets","l":"2","d":"A collection of neurons/operations that process data together, neural networks are composed of stacked layers.","n":2},{"s":"leaky-relu","t":"Leaky ReLU","r":"neural-nets","l":"2","d":"A variant of ReLU allowing small negative values (f(x) = x if x > 0, else αx where α ≈ 0.01), preventing de…","n":4},{"s":"learning-rate","t":"Learning Rate","r":"training","l":"1","d":"The hyperparameter that sets how big a step gradient descent takes on each update; too high makes training…","n":4},{"s":"learning-rate-schedule","t":"Learning Rate Schedule","a":["Learning Rate Decay"],"r":"training","l":"2","d":"A strategy for adjusting the learning rate during training (decay, warm-up, cosine annealing) to improve co…","n":5},{"s":"lemmatization","t":"Lemmatization","r":"language","l":"3","d":"Reducing words to their base or dictionary form (running → run) using linguistic knowledge.","n":2},{"s":"length-penalty","t":"Length Penalty","r":"language","l":"3","d":"Adjusting generation scores based on output length to avoid bias toward shorter or longer sequences.","n":12},{"s":"lime","t":"LIME","r":"evaluation","l":"2","d":"Local Interpretable Model-agnostic Explanations - explaining individual predictions by approximating with s…","n":3},{"s":"linear-regression","t":"Linear Regression","r":"foundations","l":"2","d":"Predicting a number as a weighted sum of input features, with the weights chosen to minimize the squared er…","n":3},{"s":"lsm","t":"Liquid State Machine","r":"neural-nets","l":"3","d":"A reservoir computing model where a recurrent network acts as a dynamic reservoir for temporal pattern reco…","n":4},{"s":"log-loss","t":"Log Loss","r":"evaluation","l":"2","d":"Logarithmic loss measuring the accuracy of probabilistic predictions, penalizing confident wrong predictions.","n":8},{"s":"lstm","t":"Long Short-Term Memory","r":"neural-nets","l":"2","d":"A type of RNN architecture with gates that can learn long-term dependencies, solving the vanishing gradient…","n":10},{"s":"lookahead","t":"Lookahead Optimizer","r":"training","l":"3","d":"A wrapper that maintains fast and slow weights, periodically updating slow weights, improving convergence.","n":9},{"s":"lora","t":"LoRA","r":"training","l":"2","d":"Low-Rank Adaptation - a parameter-efficient fine-tuning method that updates only small low-rank matrices in…","n":21},{"s":"loss-function","t":"Loss Function","r":"training","l":"1","d":"A function that scores how wrong a model's prediction is as a single number, which training then works to m…","n":2},{"s":"machine-learning","t":"Machine Learning","r":"foundations","l":"1","d":"Building systems that learn patterns from data instead of following hand-written rules, getting better at a…","n":0},{"s":"machine-translation","t":"Machine Translation","r":"language","l":"2","d":"Automatically translating text from one language to another using neural models (typically encoder-decoder…","n":4},{"s":"manifold-hypothesis","t":"Manifold Hypothesis","r":"foundations","l":"3","d":"The assumption that high-dimensional data lies on or near a lower-dimensional manifold, justifying dimensio…","n":5},{"s":"mcmc","t":"Markov Chain Monte Carlo","r":"foundations","l":"3","d":"Sampling methods for approximating distributions, especially for Bayesian inference in complex models.","n":1},{"s":"mdp","t":"Markov Decision Process","r":"agents","l":"2","d":"A mathematical framework for modeling sequential decision-making with states, actions, rewards, and transit…","n":5},{"s":"mask-r-cnn","t":"Mask R-CNN","r":"vision","l":"3","d":"Extending Faster R-CNN to instance segmentation by adding a mask prediction branch.","n":16},{"s":"masked-language-modeling","t":"Masked Language Modeling","r":"language","l":"2","d":"A pre-training objective where random tokens are masked and the model learns to predict them from context.","n":10},{"s":"masking","t":"Masking","r":"language","l":"2","d":"Hiding some positions from a model, either to control what attention may look at (causal and padding masks)…","n":10},{"s":"mcc","t":"Matthews Correlation Coefficient","r":"evaluation","l":"3","d":"A balanced measure for binary classification considering all confusion matrix values, robust to imbalance.","n":4},{"s":"map","t":"Maximum A Posteriori","r":"foundations","l":"3","d":"Parameter estimation that incorporates prior beliefs, maximizing posterior probability rather than just lik…","n":3},{"s":"mle","t":"Maximum Likelihood Estimation","r":"foundations","l":"2","d":"Finding model parameters that maximize the probability of observing the training data.","n":0},{"s":"maxout","t":"Maxout","r":"neural-nets","l":"3","d":"An activation function that outputs the maximum of multiple linear functions, providing universal approxima…","n":7},{"s":"mae","t":"Mean Absolute Error","r":"evaluation","l":"3","d":"The average absolute difference between predictions and actual values, a regression metric less sensitive t…","n":3},{"s":"mean-average-precision","t":"Mean Average Precision (mAP)","r":"evaluation","l":"2","d":"The mean of average-precision scores across classes or queries: the headline metric for object detection an…","n":8},{"s":"mse","t":"Mean Squared Error","r":"training","l":"2","d":"A loss function for regression that computes the average squared difference between predictions and targets.","n":6},{"s":"medical-diagnosis","t":"Medical Diagnosis","r":"vision","l":"3","d":"AI systems assisting healthcare professionals in diagnosing diseases from medical images, patient data, and…","n":7},{"s":"membership-inference","t":"Membership Inference","r":"shipping","l":"3","d":"Determining if a specific example was in the training dataset, a privacy concern.","n":6},{"s":"message-passing","t":"Message Passing","r":"neural-nets","l":"3","d":"The fundamental operation in GNNs where nodes exchange and aggregate information with neighbors.","n":3},{"s":"meta-learning","t":"Meta-Learning","r":"training","l":"3","d":"Learning to learn - training models that can quickly adapt to new tasks with minimal data, often applied to…","n":21},{"s":"meta-rl","t":"Meta-RL","r":"agents","l":"3","d":"Learning to adapt quickly to new RL tasks from experience on related tasks.","n":23},{"s":"metric-learning","t":"Metric Learning","r":"training","l":"2","d":"Learning a distance function, usually via an embedding, so that items that should count as similar end up c…","n":9},{"s":"mini-batch-gd","t":"Mini-Batch Gradient Descent","r":"training","l":"2","d":"Computing gradients on small batches of data, balancing SGD's noise with full-batch GD's stability.","n":5},{"s":"mish","t":"Mish","r":"neural-nets","l":"3","d":"A smooth, non-monotonic activation function (x * tanh(softplus(x))) providing better gradients than ReLU.","n":4},{"s":"mixed-precision","t":"Mixed Precision Training","a":["Automatic Mixed Precision","AMP"],"r":"training","l":"2","d":"Using lower precision (FP16) for some computations while keeping FP32 for stability, speeding up training.","n":3},{"s":"moe","t":"Mixture of Experts","r":"neural-nets","l":"2","d":"An architecture where multiple specialized sub-networks (experts) process inputs, with a gating network rou…","n":3},{"s":"mixup","t":"Mixup","r":"training","l":"3","d":"Data augmentation creating synthetic examples by interpolating between training examples and their labels.","n":4},{"s":"ml-security","t":"ML Security","r":"shipping","l":"2","d":"Protecting machine learning systems from attackers who tamper with training data, craft inputs that fool th…","n":3},{"s":"mlops","t":"MLOps","r":"shipping","l":"2","d":"Practices for deploying, monitoring, and maintaining machine learning models in production, combining ML an…","n":4},{"s":"model-caching","t":"Model Caching","r":"shipping","l":"3","d":"Storing frequently requested predictions to reduce latency and computation.","n":3},{"s":"model-card","t":"Model Card","r":"shipping","l":"3","d":"Documentation describing a model's characteristics, intended use, limitations, and ethical considerations f…","n":2},{"s":"model-compression","t":"Model Compression","r":"shipping","l":"2","d":"Techniques to reduce model size and computational requirements (quantization, pruning, distillation) for ef…","n":3},{"s":"deployment","t":"Model Deployment","r":"shipping","l":"2","d":"Getting a trained model out of the notebook and into a running system where real users or applications can…","n":3},{"s":"model-drift","t":"Model Drift","r":"shipping","l":"3","d":"Degradation of model performance over time due to changes in the relationship between features and target.","n":4},{"s":"model-endpoint","t":"Model Endpoint","r":"shipping","l":"2","d":"A deployed service exposing a model's predictions via API requests.","n":4},{"s":"model-extraction","t":"Model Extraction","r":"shipping","l":"3","d":"Stealing a model's functionality by querying it and training a copy.","n":8},{"s":"model-inversion","t":"Model Inversion","r":"shipping","l":"3","d":"Attacks that reconstruct training data or private information from model parameters or outputs.","n":3},{"s":"model-lineage","t":"Model Lineage","r":"shipping","l":"3","d":"Tracking the origin and dependencies of models including data, code, and parameters.","n":7},{"s":"model-monitoring","t":"Model Monitoring","r":"shipping","l":"2","d":"Tracking model performance, data distribution, and predictions in production to detect issues and degradation.","n":4},{"s":"model-parallelism","t":"Model Parallelism","r":"training","l":"2","d":"Splitting a single model across several devices, by layers or within layers, so that a model too large for…","n":6},{"s":"performance-degradation","t":"Model Performance Degradation","r":"shipping","l":"2","d":"Decline in model quality over time due to distribution shift or changing patterns.","n":5},{"s":"model-registry","t":"Model Registry","r":"shipping","l":"2","d":"A centralized repository for tracking, versioning, and managing trained models.","n":5},{"s":"model-reproducibility","t":"Model Reproducibility","r":"shipping","l":"2","d":"The ability to recreate exact model results given the same code, data, and environment.","n":2},{"s":"model-retraining","t":"Model Retraining","r":"shipping","l":"2","d":"Periodically updating models with new data to maintain performance as distributions change.","n":5},{"s":"model-serving","t":"Model Serving","r":"shipping","l":"2","d":"Deploying trained models as services that can handle prediction requests in production environments.","n":3},{"s":"model-versioning","t":"Model Versioning","r":"shipping","l":"2","d":"Tracking different versions of models to enable reproducibility and rollback.","n":5},{"s":"model-watermarking","t":"Model Watermarking","r":"shipping","l":"3","d":"Embedding identifiable signatures in models to prove ownership and detect theft.","n":2},{"s":"model-based-rl","t":"Model-Based RL","r":"agents","l":"2","d":"Reinforcement learning using learned environment models for planning and improving sample efficiency.","n":5},{"s":"model-free-rl","t":"Model-Free RL","r":"agents","l":"2","d":"Reinforcement learning directly learning policies or value functions without modeling environment dynamics.","n":2},{"s":"momentum","t":"Momentum","r":"training","l":"2","d":"An optimization technique that accelerates gradient descent by accumulating past gradients, helping escape…","n":5},{"s":"mcts","t":"Monte Carlo Tree Search","r":"agents","l":"3","d":"A search algorithm combining tree search with random sampling, used in game-playing AIs.","n":3},{"s":"multi-agent-rl","t":"Multi-Agent RL","r":"agents","l":"3","d":"Reinforcement learning with multiple agents that interact and potentially cooperate or compete.","n":3},{"s":"multi-armed-bandit","t":"Multi-Armed Bandit","r":"agents","l":"2","d":"The simplest decision problem with a trade-off between exploring and exploiting: repeatedly pick one of sev…","n":3},{"s":"multi-head-attention","t":"Multi-Head Attention","r":"language","l":"2","d":"Running multiple attention operations in parallel with different learned projections, capturing diverse rel…","n":9},{"s":"mlp","t":"Multi-Layer Perceptron","r":"neural-nets","l":"2","d":"A feedforward neural network with multiple layers of perceptrons, capable of learning non-linear functions.","n":5},{"s":"multi-task-learning","t":"Multi-Task Learning","r":"training","l":"3","d":"Training a single model on multiple related tasks simultaneously to improve generalization and efficiency.","n":7},{"s":"multimodal-model","t":"Multimodal Model","a":["Multimodal Learning"],"r":"vision","l":"2","d":"Models processing multiple data types (text, images, audio) jointly, like GPT-4V, Gemini, or CLIP.","n":19},{"s":"mutual-information","t":"Mutual Information","r":"foundations","l":"3","d":"A measure of dependence between variables, quantifying how much knowing one reduces uncertainty about the o…","n":1},{"s":"n-gram","t":"N-gram","r":"language","l":"3","d":"A contiguous sequence of n items (words, characters) from text, used in language modeling and feature extra…","n":1},{"s":"ner","t":"Named Entity Recognition","r":"language","l":"2","d":"Identifying and classifying named entities (people, organizations, locations) in text into predefined categ…","n":2},{"s":"nli","t":"Natural Language Inference","r":"language","l":"3","d":"Determining logical relationships (entailment, contradiction, neutral) between sentence pairs.","n":2},{"s":"nlp","t":"Natural Language Processing","r":"language","l":"1","d":"The field of AI that lets computers read, interpret, translate and generate human language, from spam filte…","n":0},{"s":"nesterov-momentum","t":"Nesterov Momentum","r":"training","l":"3","d":"A momentum variant that looks ahead before computing gradients, often converging faster.","n":6},{"s":"nas","t":"Neural Architecture Search","r":"training","l":"3","d":"Automated methods for discovering optimal neural network architectures, using techniques like reinforcement…","n":9},{"s":"neural-network","t":"Neural Network","r":"neural-nets","l":"1","d":"A model built from layers of simple units, each taking a weighted sum of its inputs and applying a nonlinea…","n":1},{"s":"neural-ode","t":"Neural ODE","r":"neural-nets","l":"3","d":"Neural Ordinary Differential Equations - modeling continuous-depth networks as ODEs, enabling adaptive comp…","n":10},{"s":"scaling-laws","t":"Neural Scaling Laws","r":"language","l":"2","d":"Empirical relationships showing how model performance improves predictably with model size, data, and compute.","n":19},{"s":"neuromorphic-computing","t":"Neuromorphic Computing","r":"shipping","l":"3","d":"Hardware and algorithms designed to mimic the brain's structure and function, enabling efficient spike-base…","n":3},{"s":"next-token-prediction","t":"Next-Token Prediction","r":"language","l":"2","d":"The task of predicting the next token in a sequence from all the tokens before it: the training objective b…","n":4},{"s":"no-free-lunch","t":"No Free Lunch Theorem","r":"foundations","l":"3","d":"The principle that no single ML algorithm works best for all problems - algorithm choice depends on the spe…","n":0},{"s":"node-embedding","t":"Node Embedding","r":"neural-nets","l":"3","d":"Learning vector representations of graph nodes that capture structural and feature information.","n":7},{"s":"nms","t":"Non-Maximum Suppression","r":"vision","l":"3","d":"Filtering overlapping detection boxes by keeping only the most confident predictions.","n":7},{"s":"normalization","t":"Normalization","r":"foundations","l":"2","d":"Scaling features to a standard range (typically 0-1 using min-max scaling) to improve model training and co…","n":2},{"s":"object-detection","t":"Object Detection","r":"vision","l":"1","d":"Finding every object of interest in an image and giving each a class label, a confidence score and a boundi…","n":5},{"s":"occams-razor","t":"Occam's Razor","r":"foundations","l":"3","d":"The principle that simpler models should be preferred when they perform equally well, reducing overfitting.","n":3},{"s":"offline-rl","t":"Offline RL","r":"agents","l":"3","d":"Learning policies from fixed datasets without environment interaction, enabling learning from logs.","n":4},{"s":"one-hot-encoding","t":"One-Hot Encoding","r":"foundations","l":"3","d":"Converting categorical variables into binary vectors with one element set to 1 and others to 0.","n":2},{"s":"online-learning","t":"Online Learning","r":"training","l":"3","d":"Models that learn continuously from streaming data, updating incrementally as new data arrives.","n":2},{"s":"ocr","t":"Optical Character Recognition","r":"vision","l":"3","d":"Converting images of text (scanned documents, photos) into machine-readable text using computer vision.","n":4},{"s":"optical-flow","t":"Optical Flow","r":"vision","l":"3","d":"Estimating motion patterns between video frames by tracking pixel movements.","n":4},{"s":"optimizer","t":"Optimizer","r":"training","l":"2","d":"The algorithm that turns gradients into weight updates during training, deciding how far and in which direc…","n":4},{"s":"ood","t":"Out-of-Distribution","r":"evaluation","l":"2","d":"Data that differs significantly from the training distribution, where models often perform poorly or unreli…","n":2},{"s":"overfitting","t":"Overfitting","r":"training","l":"1","d":"When a model fits its training data too closely, noise included, so it scores well on examples it has seen…","n":2},{"s":"pac-learning","t":"PAC Learning","r":"foundations","l":"3","d":"Probably Approximately Correct - a theoretical framework for analyzing learning algorithm guarantees.","n":7},{"s":"padding","t":"Padding","r":"neural-nets","l":"3","d":"Adding borders of zeros (or other values) around input to control output spatial dimensions in convolutions.","n":3},{"s":"padding-token","t":"Padding Token","r":"language","l":"3","d":"A special token used to make sequences the same length in a batch, typically ignored during computation.","n":3},{"s":"parameter","t":"Parameter","r":"neural-nets","l":"1","d":"A number inside a model, a weight or a bias, that training adjusts to reduce the loss. Together, a model's…","n":2},{"s":"peft","t":"Parameter-Efficient Fine-Tuning (PEFT)","r":"training","l":"2","d":"A family of methods that adapt a large pretrained model by training a small number of new or selected param…","n":9},{"s":"paraphrase-detection","t":"Paraphrase Detection","r":"language","l":"3","d":"Determining if two text segments express the same meaning in different words.","n":8},{"s":"pos-tagging","t":"Part-of-Speech Tagging","r":"language","l":"3","d":"Labeling words in text with their grammatical roles (noun, verb, adjective, etc.).","n":2},{"s":"perceptron","t":"Perceptron","r":"neural-nets","l":"2","d":"The simplest neural network unit, a single-layer binary classifier that inspired modern deep learning.","n":2},{"s":"perplexity","t":"Perplexity","r":"evaluation","l":"2","d":"A metric measuring how well a language model predicts text - lower perplexity indicates better prediction.","n":12},{"s":"personalization","t":"Personalization","r":"foundations","l":"3","d":"Tailoring content, recommendations, or experiences to individual users based on their preferences and behav…","n":3},{"s":"pipeline-parallelism","t":"Pipeline Parallelism","r":"training","l":"3","d":"Splitting model layers across devices and processing micro-batches in pipeline fashion.","n":9},{"s":"policy","t":"Policy","r":"agents","l":"1","d":"The rule an RL agent uses to choose actions: a mapping from each state to an action, or to a probability di…","n":3},{"s":"policy-gradient","t":"Policy Gradient","r":"agents","l":"2","d":"RL methods that directly optimize the policy by computing gradients of expected reward with respect to poli…","n":8},{"s":"pooling","t":"Pooling","r":"neural-nets","l":"2","d":"A down-sampling operation in CNNs that reduces spatial dimensions while retaining important features (max p…","n":4},{"s":"pose-estimation","t":"Pose Estimation","r":"vision","l":"3","d":"Detecting body keypoints to determine human pose from images or video.","n":4},{"s":"positional-encoding","t":"Positional Encoding","r":"language","l":"2","d":"Adding position information to token embeddings so the model understands word order in sequences.","n":9},{"s":"posterior","t":"Posterior Distribution","r":"foundations","l":"3","d":"In Bayesian methods, the updated belief about parameters after observing data.","n":1},{"s":"ppo","t":"PPO","r":"agents","l":"2","d":"Proximal Policy Optimization - a stable and efficient policy gradient algorithm widely used in RLHF for tra…","n":12},{"s":"pre-training","t":"Pre-training","r":"training","l":"1","d":"Training a model on a large dataset (often self-supervised) before fine-tuning on specific tasks, enabling…","n":5},{"s":"precision","t":"Precision","r":"evaluation","l":"1","d":"Of everything a model flagged as positive, the fraction that really was positive: TP / (TP + FP). It answer…","n":4},{"s":"pr-curve","t":"Precision-Recall Curve","r":"evaluation","l":"2","d":"A curve showing the tradeoff between precision and recall at different thresholds.","n":6},{"s":"prediction-confidence","t":"Prediction Confidence","r":"shipping","l":"2","d":"A measure of model certainty in its predictions, important for reliability and user trust.","n":4},{"s":"prelu","t":"PReLU","r":"neural-nets","l":"3","d":"Parametric ReLU - like Leaky ReLU but with learnable negative slope parameter.","n":5},{"s":"pca","t":"Principal Component Analysis","r":"foundations","l":"2","d":"A dimensionality reduction technique that transforms data into orthogonal components ordered by variance ex…","n":6},{"s":"prior","t":"Prior Distribution","r":"foundations","l":"3","d":"In Bayesian methods, the initial belief about parameters before observing data.","n":1},{"s":"privacy-preserving-ml","t":"Privacy-Preserving ML","r":"shipping","l":"2","d":"Techniques for training and deploying models while protecting individual privacy (federated learning, diffe…","n":2},{"s":"prompt-engineering","t":"Prompt Engineering","r":"language","l":"1","d":"Designing a model's input (instructions, context, examples and output format) to get better results without…","n":18},{"s":"prompt-template","t":"Prompt Template","r":"language","l":"3","d":"A reusable structure for crafting prompts with placeholders for variables, improving consistency.","n":19},{"s":"prompt-tuning","t":"Prompt Tuning","r":"language","l":"3","d":"Learning continuous prompt embeddings while keeping the LLM frozen, an efficient alternative to fine-tuning.","n":22},{"s":"protein-folding","t":"Protein Folding","r":"foundations","l":"3","d":"Predicting 3D protein structures from amino acid sequences, revolutionized by AlphaFold.","n":3},{"s":"pruning","t":"Pruning","r":"shipping","l":"2","d":"Removing unnecessary weights or neurons from a trained model to reduce size and computation while maintaini…","n":4},{"s":"q-learning","t":"Q-Learning","r":"agents","l":"2","d":"A model-free RL algorithm that learns action-value functions (Q-values) to determine optimal actions in eac…","n":6},{"s":"quantization","t":"Quantization","r":"shipping","l":"2","d":"Reducing model precision (FP32 → INT8) to decrease size and increase inference speed with minimal accuracy…","n":6},{"s":"query-key-value","t":"Query-Key-Value","r":"language","l":"2","d":"The three learned projections in attention mechanisms used to compute attention weights and outputs.","n":8},{"s":"question-answering","t":"Question Answering","r":"language","l":"2","d":"Systems that automatically answer questions posed in natural language, often by reading and comprehending t…","n":1},{"s":"r-cnn","t":"R-CNN","r":"vision","l":"3","d":"Region-based CNN - an object detection approach using selective search and CNN features.","n":8},{"s":"r-squared","t":"R-squared","r":"evaluation","l":"3","d":"Coefficient of determination - measures the proportion of variance in the target variable explained by the…","n":3},{"s":"rademacher-complexity","t":"Rademacher Complexity","r":"foundations","l":"3","d":"A measure of how well a model class can fit random noise, indicating capacity and generalization ability.","n":9},{"s":"rbf-network","t":"Radial Basis Function Network","r":"neural-nets","l":"3","d":"A neural network using radial basis functions as activation functions, useful for function approximation an…","n":2},{"s":"random-crop","t":"Random Crop","r":"vision","l":"3","d":"Extracting random patches from images for augmentation and training.","n":9},{"s":"random-forest","t":"Random Forest","r":"foundations","l":"2","d":"An ensemble of decision trees trained on random subsets of data and features, reducing overfitting through…","n":6},{"s":"reading-comprehension","t":"Reading Comprehension","r":"language","l":"3","d":"Answering questions about a text passage, testing understanding.","n":2},{"s":"recall","t":"Recall","a":["Sensitivity"],"r":"evaluation","l":"1","d":"The share of actual positives a model correctly identifies, TP / (TP + FN); also called sensitivity or the…","n":4},{"s":"receptive-field","t":"Receptive Field","r":"neural-nets","l":"3","d":"The region of input that influences a particular neuron's output, growing larger in deeper layers of CNNs.","n":3},{"s":"recommender-system","t":"Recommender System","r":"foundations","l":"2","d":"AI systems that suggest items (products, content) to users based on preferences, behavior, and similarity.","n":2},{"s":"rnn","t":"Recurrent Neural Network","r":"neural-nets","l":"1","d":"A neural network architecture with loops that allow information to persist, designed for sequential data li…","n":2},{"s":"rpn","t":"Region Proposal Network","r":"vision","l":"3","d":"A network generating candidate object locations for two-stage detectors like Faster R-CNN.","n":9},{"s":"regression","t":"Regression","r":"foundations","l":"1","d":"A supervised learning task where the model predicts continuous numerical values rather than discrete catego…","n":2},{"s":"regularization","t":"Regularization","r":"training","l":"1","d":"Any change to training meant to make a model generalize better rather than fit its training data more close…","n":3},{"s":"reinforcement-learning","t":"Reinforcement Learning","r":"agents","l":"1","d":"Learning through interaction with an environment, receiving rewards or penalties to learn optimal behavior…","n":1},{"s":"relation-extraction","t":"Relation Extraction","r":"language","l":"3","d":"Identifying semantic relationships between entities in text.","n":3},{"s":"release-strategy","t":"Release Strategy","r":"shipping","l":"2","d":"The plan for how a new model or software version reaches users: all at once, to a small slice first, or sil…","n":4},{"s":"relu","t":"ReLU","r":"neural-nets","l":"2","d":"Rectified Linear Unit - an activation function that outputs the input if positive, zero otherwise. f(x) = m…","n":3},{"s":"repetition-penalty","t":"Repetition Penalty","r":"language","l":"3","d":"A technique reducing the likelihood of previously generated tokens to avoid repetitive outputs.","n":5},{"s":"representation-learning","t":"Representation Learning","r":"neural-nets","l":"1","d":"Letting a model learn its own internal description of the data, such as embeddings or layered visual featur…","n":5},{"s":"request-batching","t":"Request Batching","r":"shipping","l":"3","d":"Combining multiple inference requests into batches to improve throughput.","n":4},{"s":"reservoir-computing","t":"Reservoir Computing","r":"neural-nets","l":"2","d":"A way to use a recurrent network without training it: a fixed random reservoir turns an input signal into r…","n":3},{"s":"residual-connection","t":"Residual Connection","r":"neural-nets","l":"2","d":"Skip connections that allow gradients to flow directly through a network, enabling training of very deep ne…","n":9},{"s":"resnet","t":"ResNet","r":"vision","l":"2","d":"Residual Network - a CNN architecture using skip connections to enable training of very deep networks (up t…","n":12},{"s":"rest-api","t":"REST API","r":"shipping","l":"3","d":"A web service interface commonly used for serving model predictions over HTTP.","n":4},{"s":"rbm","t":"Restricted Boltzmann Machine","r":"neural-nets","l":"3","d":"A simpler variant of Boltzmann machines with no intra-layer connections, used for unsupervised learning and…","n":7},{"s":"retinanet","t":"RetinaNet","r":"vision","l":"3","d":"A single-stage object detector using focal loss to handle class imbalance.","n":19},{"s":"rag","t":"Retrieval-Augmented Generation","r":"language","l":"1","d":"Fetching relevant documents at question time and adding them to an LLM's prompt, so answers can draw on cur…","n":18},{"s":"rig","t":"Retrieval-Interleaved Generation","r":"language","l":"3","d":"Dynamically retrieving information during generation rather than just before, allowing models to gather fac…","n":19},{"s":"reward","t":"Reward","r":"agents","l":"1","d":"The single number an environment sends back after each action, telling a reinforcement learning agent how g…","n":2},{"s":"rlhf","t":"RLHF","r":"language","l":"2","d":"Reinforcement Learning from Human Feedback - training models using human preferences to align behavior with…","n":20},{"s":"rmsprop","t":"RMSprop","r":"training","l":"3","d":"An optimizer using moving average of squared gradients to adapt learning rates, addressing AdaGrad's dimini…","n":6},{"s":"robustness","t":"Robustness","r":"evaluation","l":"2","d":"A model's ability to maintain performance under distribution shifts, adversarial attacks, or noisy inputs.","n":3},{"s":"roc-curve","t":"ROC Curve","r":"evaluation","l":"2","d":"Receiver Operating Characteristic curve - plots true positive rate vs false positive rate at various classi…","n":8},{"s":"rouge","t":"ROUGE Score","r":"evaluation","l":"3","d":"Metrics for evaluating text summarization by measuring overlap of n-grams, word sequences, and word pairs w…","n":7},{"s":"saliency-map","t":"Saliency Map","r":"evaluation","l":"2","d":"A visualization highlighting input regions most important for model predictions.","n":8},{"s":"scaled-dot-product","t":"Scaled Dot-Product Attention","r":"language","l":"3","d":"The attention computation using dot product of queries and keys, scaled by dimension to stabilize gradients.","n":11},{"s":"smpc","t":"Secure Multi-Party Computation","r":"shipping","l":"3","d":"Protocols allowing parties to jointly compute functions while keeping inputs private.","n":3},{"s":"self-attention","t":"Self-Attention","r":"language","l":"1","d":"An attention step in which every token in a sequence looks at every token in that same sequence and builds…","n":8},{"s":"self-supervised-learning","t":"Self-Supervised Learning","r":"training","l":"1","d":"Training on unlabeled data by making the data supply its own answers: hide or alter part of an input and ha…","n":2},{"s":"srl","t":"Semantic Role Labeling","r":"language","l":"3","d":"Identifying the semantic relationships between predicates and their arguments in sentences.","n":4},{"s":"semantic-search","t":"Semantic Search","r":"language","l":"2","d":"Search that matches on meaning rather than exact words, by turning queries and documents into embeddings an…","n":8},{"s":"semantic-segmentation","t":"Semantic Segmentation","r":"vision","l":"1","d":"Labelling every pixel in an image with a class, such as road, car or sky, so the output is a map of what is…","n":6},{"s":"semantic-similarity","t":"Semantic Similarity","r":"language","l":"2","d":"Measuring how similar two pieces of text are in meaning, often using embedding-based distance metrics.","n":7},{"s":"semi-supervised-learning","t":"Semi-Supervised Learning","r":"foundations","l":"2","d":"Learning from a combination of labeled and unlabeled data, leveraging abundant unlabeled data to improve pe…","n":3},{"s":"sentencepiece","t":"SentencePiece","r":"language","l":"3","d":"A language-agnostic tokenization library that treats text as a sequence of Unicode characters.","n":3},{"s":"sentiment-analysis","t":"Sentiment Analysis","r":"language","l":"2","d":"Determining the emotional tone or opinion expressed in text (positive, negative, neutral).","n":5},{"s":"seq2seq","t":"Sequence-to-Sequence","r":"language","l":"2","d":"Models that transform input sequences to output sequences, used for translation, summarization, and generat…","n":3},{"s":"shadow-deployment","t":"Shadow Deployment","r":"shipping","l":"3","d":"Running a new model alongside production without affecting user experience, for validation.","n":5},{"s":"shap","t":"SHAP","r":"evaluation","l":"2","d":"SHapley Additive exPlanations - a unified approach to explaining model predictions using game theory.","n":3},{"s":"sharded-data-parallel","t":"Sharded Data Parallelism","r":"training","l":"3","d":"Distributing model states across devices to train models larger than single-device memory.","n":5},{"s":"shot","t":"Shot","r":"language","l":"2","d":"An example provided in a prompt - zero-shot (no examples), few-shot (a few examples), one-shot (one example).","n":19},{"s":"sigmoid","t":"Sigmoid","r":"neural-nets","l":"2","d":"An activation function that squashes values to range (0,1), often used for binary classification and gates…","n":3},{"s":"sim-to-real","t":"Sim-to-Real Transfer","r":"agents","l":"3","d":"Transferring policies trained in simulation to real-world deployment, crucial for robotics.","n":8},{"s":"skip-connection","t":"Skip Connection","r":"neural-nets","l":"3","d":"Direct connections bypassing one or more layers, helping gradient flow and enabling deeper networks.","n":3},{"s":"slot-filling","t":"Slot Filling","r":"language","l":"3","d":"Extracting specific pieces of information (slots) needed to fulfill a user's intent.","n":8},{"s":"softmax","t":"Softmax","r":"neural-nets","l":"1","d":"A function that turns a list of scores (logits) into probabilities that are all positive and sum to 1; the…","n":3},{"s":"special-token","t":"Special Token","r":"language","l":"3","d":"Reserved tokens with special meanings like [CLS], [SEP], [MASK], [PAD] used in various model architectures.","n":2},{"s":"specificity","t":"Specificity","r":"evaluation","l":"2","d":"The proportion of actual negatives correctly identified.","n":6},{"s":"spectral-normalization","t":"Spectral Normalization","r":"neural-nets","l":"3","d":"Constraining the spectral norm of weight matrices to stabilize GAN training.","n":8},{"s":"speech-recognition","t":"Speech Recognition","r":"language","l":"2","d":"Converting spoken language into text using acoustic models and language models, now dominated by deep learn…","n":5},{"s":"snn","t":"Spiking Neural Network","r":"neural-nets","l":"3","d":"Networks inspired by biological neurons that communicate through discrete spikes, incorporating temporal dy…","n":2},{"s":"squeeze-excitation","t":"Squeeze-and-Excitation","r":"vision","l":"3","d":"A channel attention mechanism that adaptively recalibrates channel-wise feature responses, improving CNNs.","n":10},{"s":"stable-diffusion","t":"Stable Diffusion","r":"vision","l":"2","d":"A latent diffusion model for text-to-image generation that operates in compressed latent space for efficiency.","n":9},{"s":"standardization","t":"Standardization","r":"foundations","l":"2","d":"Rescaling each feature to mean 0 and standard deviation 1, so features measured in different units are on a…","n":2},{"s":"stemming","t":"Stemming","r":"language","l":"3","d":"Reducing words to their root form by removing affixes (prefixes, suffixes, infixes), simpler than lemmatiza…","n":2},{"s":"step-decay","t":"Step Decay","r":"training","l":"3","d":"Reducing learning rate by a factor at specific epochs, a simple scheduling strategy.","n":6},{"s":"sgd","t":"Stochastic Gradient Descent","r":"training","l":"2","d":"A variant of gradient descent that updates parameters using gradients computed on a single random training…","n":5},{"s":"stop-words","t":"Stop Words","r":"language","l":"3","d":"Common words (the, is, at) often removed in NLP preprocessing as they carry little semantic meaning.","n":2},{"s":"stride","t":"Stride","r":"neural-nets","l":"3","d":"The step size by which a convolutional filter or pooling window moves across the input.","n":3},{"s":"student-model","t":"Student Model","r":"training","l":"3","d":"The smaller model in knowledge distillation learning to mimic the teacher's behavior.","n":7},{"s":"style-transfer","t":"Style Transfer","r":"vision","l":"3","d":"Transferring artistic style from one image to another while preserving content.","n":7},{"s":"subword-tokenization","t":"Subword Tokenization","r":"language","l":"2","d":"Breaking words into smaller units, balancing vocabulary size with representation granularity.","n":2},{"s":"super-resolution","t":"Super-Resolution","r":"vision","l":"3","d":"Enhancing image resolution using deep learning to recover high-frequency details.","n":5},{"s":"supervised-learning","t":"Supervised Learning","r":"foundations","l":"1","d":"Learning from examples paired with the correct answer, so a model can predict answers for new inputs it has…","n":1},{"s":"svm","t":"Support Vector Machine","r":"foundations","l":"2","d":"A supervised learning algorithm that finds the optimal hyperplane to separate classes with maximum margin.","n":3},{"s":"swish","t":"Swish","r":"neural-nets","l":"3","d":"A smooth activation function (x * sigmoid(x)) that often outperforms ReLU, discovered through neural archit…","n":5},{"s":"synthetic-data","t":"Synthetic Data","r":"foundations","l":"3","d":"Artificially generated data created to augment training sets, protect privacy, or simulate rare scenarios.","n":1},{"s":"system-prompt","t":"System Prompt","r":"language","l":"2","d":"Initial instructions defining the model's role, behavior, and constraints for the conversation.","n":18},{"s":"t5","t":"T5","r":"language","l":"2","d":"Text-to-Text Transfer Transformer - frames all NLP tasks as text-to-text problems using a unified encoder-d…","n":12},{"s":"teacher-model","t":"Teacher Model","r":"training","l":"3","d":"The larger, more accurate model in knowledge distillation that guides student training.","n":7},{"s":"temperature","t":"Temperature","a":["Softmax Temperature"],"r":"language","l":"2","d":"A sampling parameter controlling randomness in generation - lower values make output more deterministic, hi…","n":9},{"s":"tensor-parallelism","t":"Tensor Parallelism","r":"training","l":"3","d":"Splitting individual layers/tensors across devices for very large models.","n":9},{"s":"test-set","t":"Test Set","r":"evaluation","l":"1","d":"Data locked away until the end of a project and used once to estimate how the finished model will perform o…","n":2},{"s":"text-classification","t":"Text Classification","r":"language","l":"2","d":"Assigning categories or labels to text documents, a fundamental NLP task.","n":4},{"s":"text-generation","t":"Text Generation","r":"language","l":"2","d":"Automatically creating coherent text using language models, from simple completion to creative writing.","n":4},{"s":"text-summarization","t":"Text Summarization","r":"language","l":"3","d":"Generating concise summaries of longer texts, either extractive (selecting sentences) or abstractive (gener…","n":4},{"s":"text-to-speech","t":"Text-to-Speech","r":"vision","l":"3","d":"Synthesizing natural-sounding speech from text, using neural vocoders and attention-based models.","n":4},{"s":"textual-entailment","t":"Textual Entailment","r":"language","l":"3","d":"Determining if one text fragment logically follows from another.","n":3},{"s":"tf-idf","t":"TF-IDF","r":"language","l":"3","d":"Term Frequency-Inverse Document Frequency - a statistical measure of word importance in documents, used for…","n":4},{"s":"throughput","t":"Throughput","r":"shipping","l":"3","d":"The number of predictions or tokens a model can process per unit of time, a key deployment performance metric.","n":3},{"s":"time-series-forecasting","t":"Time Series Forecasting","r":"foundations","l":"3","d":"Predicting future values based on historical sequential data, using models like ARIMA, LSTMs, or Transformers.","n":3},{"s":"token","t":"Token","r":"language","l":"1","d":"The unit of text a language model reads and writes: a word, part of a word, or a character, drawn from the…","n":0},{"s":"tokenization","t":"Tokenization","r":"language","l":"1","d":"Splitting text into tokens, usually subword pieces, and mapping each to an integer ID so a language model c…","n":1},{"s":"tool-use","t":"Tool Use","r":"agents","l":"1","d":"A language model asking the application around it to run a function, such as search, a calculator or an API…","n":18},{"s":"top-k","t":"Top-k Sampling","r":"language","l":"2","d":"A generation strategy that samples from only the k most likely next tokens, balancing quality and diversity.","n":11},{"s":"top-p","t":"Top-p Sampling","a":["Nucleus Sampling"],"r":"language","l":"3","d":"Nucleus sampling - selecting from the smallest set of tokens whose cumulative probability exceeds p, provid…","n":12},{"s":"topic-modeling","t":"Topic Modeling","r":"language","l":"3","d":"Discovering abstract topics in document collections, often using techniques like LDA.","n":4},{"s":"tpu","t":"TPU","r":"shipping","l":"3","d":"Tensor Processing Unit - Google's custom hardware accelerator designed specifically for machine learning wo…","n":1},{"s":"train-test-split","t":"Train-Test Split","r":"evaluation","l":"2","d":"Dividing a dataset into separate portions for training the model and evaluating its performance on unseen d…","n":1},{"s":"training","t":"Training","r":"training","l":"1","d":"The process of fitting a model to data by repeatedly measuring how wrong its outputs are and adjusting its…","n":1},{"s":"training-data","t":"Training Data","r":"foundations","l":"1","d":"The examples a model learns its weights from, kept separate from the validation and test data used to check…","n":1},{"s":"transfer-learning","t":"Transfer Learning","r":"training","l":"2","d":"Leveraging knowledge learned from one task/domain to improve performance on a related task with less data.","n":6},{"s":"transfer-learning-vision","t":"Transfer Learning in Vision","r":"vision","l":"2","d":"Using pre-trained vision models (ImageNet) as feature extractors or fine-tuning for specific visual tasks.","n":13},{"s":"transformer","t":"Transformer","r":"language","l":"1","d":"A neural network architecture, introduced in 2017, built from stacked self-attention and feed-forward layer…","n":8},{"s":"transposed-convolution","t":"Transposed Convolution","r":"neural-nets","l":"3","d":"An operation that upsamples feature maps, often used in decoders and generative models (also called deconvo…","n":3},{"s":"triplet-loss","t":"Triplet Loss","r":"training","l":"3","d":"A loss for learning embeddings by pulling similar examples together and pushing dissimilar ones apart.","n":9},{"s":"true-negative","t":"True Negative","r":"evaluation","l":"2","d":"Correctly predicted negative cases in classification.","n":4},{"s":"true-positive","t":"True Positive","r":"evaluation","l":"2","d":"Correctly predicted positive cases in classification.","n":4},{"s":"tee","t":"Trusted Execution Environment","r":"shipping","l":"3","d":"Secure hardware areas for protected computation, used for private AI inference.","n":3},{"s":"type-i-ii-errors","t":"Type I and Type II Errors","r":"evaluation","l":"2","d":"The two ways a yes-or-no decision can be wrong: a Type I error is a false alarm (false positive), a Type II…","n":4},{"s":"u-net","t":"U-Net","r":"vision","l":"2","d":"A CNN architecture with encoder-decoder structure and skip connections, widely used for image segmentation…","n":14},{"s":"underfitting","t":"Underfitting","r":"training","l":"2","d":"When a model is too simple to capture the underlying pattern in data, performing poorly on both training an…","n":2},{"s":"universal-approximation","t":"Universal Approximation Theorem","r":"neural-nets","l":"3","d":"The theorem stating neural networks with one hidden layer can approximate any continuous function.","n":4},{"s":"unsupervised-learning","t":"Unsupervised Learning","r":"foundations","l":"1","d":"Learning from unlabeled data to discover hidden patterns, structures, or relationships without explicit tar…","n":1},{"s":"user-prompt","t":"User Prompt","r":"language","l":"3","d":"The input provided by the user to the language model, containing questions or instructions.","n":18},{"s":"validation-set","t":"Validation Set","r":"evaluation","l":"1","d":"Data held back from training and used to choose hyperparameters and checkpoints, so those choices are judge…","n":2},{"s":"value-function","t":"Value Function","r":"agents","l":"2","d":"A function estimating expected cumulative reward from a state (state-value) or state-action pair (action-va…","n":5},{"s":"vanishing-gradient","t":"Vanishing Gradient","r":"training","l":"2","d":"A problem where gradients become extremely small during backpropagation, preventing deep networks from lear…","n":8},{"s":"vae","t":"Variational Autoencoder","r":"neural-nets","l":"2","d":"A generative model that learns a probabilistic latent space, allowing sampling of new data points similar t…","n":6},{"s":"variational-inference","t":"Variational Inference","r":"foundations","l":"3","d":"Approximating complex distributions by optimizing over a simpler family, an alternative to MCMC.","n":3},{"s":"vc-dimension","t":"VC Dimension","r":"foundations","l":"3","d":"A measure of model capacity - the largest set of points a model can shatter (classify in all possible ways).","n":8},{"s":"vector-database","t":"Vector Database","r":"language","l":"2","d":"A database optimized for storing and searching high-dimensional vectors (embeddings), enabling semantic sea…","n":7},{"s":"vgg","t":"VGG","r":"vision","l":"3","d":"A CNN architecture known for its simplicity, using small 3x3 convolutions stacked deeply, influential in co…","n":4},{"s":"vit","t":"Vision Transformer","r":"vision","l":"2","d":"Applying the transformer architecture to computer vision by treating image patches as tokens, achieving sta…","n":10},{"s":"vocabulary-size","t":"Vocabulary Size","r":"language","l":"3","d":"The number of distinct tokens a language model can process, typically 30K-100K+ tokens.","n":2},{"s":"warmup","t":"Warmup","r":"training","l":"3","d":"Gradually increasing the learning rate at training start to stabilize optimization.","n":6},{"s":"wasserstein-distance","t":"Wasserstein Distance","r":"foundations","l":"3","d":"Earth Mover's Distance - a metric measuring the minimum cost to transform one distribution into another.","n":0},{"s":"weight","t":"Weight","r":"neural-nets","l":"2","d":"Learnable parameters connecting neurons in neural networks, determining the strength of connections.","n":2},{"s":"weight-decay","t":"Weight Decay","r":"training","l":"2","d":"A regularization technique that shrinks weights toward zero during optimization. Equivalent to L2 regulariz…","n":4},{"s":"weight-initialization","t":"Weight Initialization","r":"neural-nets","l":"2","d":"How a network's weights are set before training starts: random values scaled so that signals and gradients…","n":3},{"s":"weight-normalization","t":"Weight Normalization","r":"neural-nets","l":"3","d":"Reparameterizing weight vectors to improve optimization by decoupling magnitude from direction.","n":6},{"s":"wsd","t":"Word Sense Disambiguation","r":"language","l":"3","d":"Determining which meaning of a word is used in a particular context.","n":1},{"s":"word2vec","t":"Word2Vec","r":"language","l":"2","d":"A technique for learning word embeddings that capture semantic relationships (Skip-gram and CBOW models).","n":7},{"s":"wordpiece","t":"WordPiece","r":"language","l":"3","d":"A subword tokenization algorithm used by BERT, similar to BPE but with different merging criteria.","n":3},{"s":"world-model","t":"World Model","r":"agents","l":"3","d":"A learned model of environment dynamics that can predict future states, used in model-based RL.","n":4},{"s":"xavier-initialization","t":"Xavier Initialization","r":"neural-nets","l":"3","d":"A weight initialization strategy maintaining variance across layers, improving training of deep networks.","n":5},{"s":"yolo","t":"YOLO","r":"vision","l":"2","d":"You Only Look Once - a real-time object detection architecture treating detection as regression.","n":8},{"s":"zero","t":"ZeRO","r":"training","l":"3","d":"Zero Redundancy Optimizer - techniques for memory-efficient distributed training by partitioning optimizer…","n":11},{"s":"zero-shot","t":"Zero-Shot Learning","r":"language","l":"2","d":"A model's ability to perform tasks it wasn't explicitly trained on, using only instructions or descriptions.","n":7}]