<?xml version="1.0" encoding="UTF-8"?><urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9" xmlns:news="http://www.google.com/schemas/sitemap-news/0.9" xmlns:xhtml="http://www.w3.org/1999/xhtml" xmlns:image="http://www.google.com/schemas/sitemap-image/1.1" xmlns:video="http://www.google.com/schemas/sitemap-video/1.1"><url><loc>https://llmdeepdive.com/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/"/></url><url><loc>https://llmdeepdive.com/explore/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/explore/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/explore/"/></url><url><loc>https://llmdeepdive.com/lessons/0-orientation-and-foundations/0.1-what-is-a-language-model/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/0-orientation-and-foundations/0.1-what-is-a-language-model/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/0-orientation-and-foundations/0.1-what-is-a-language-model/"/></url><url><loc>https://llmdeepdive.com/lessons/0-orientation-and-foundations/0.2-history-ngram-to-gpt/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/0-orientation-and-foundations/0.2-history-ngram-to-gpt/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/0-orientation-and-foundations/0.2-history-ngram-to-gpt/"/></url><url><loc>https://llmdeepdive.com/lessons/0-orientation-and-foundations/0.3-probability-you-need/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/0-orientation-and-foundations/0.3-probability-you-need/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/0-orientation-and-foundations/0.3-probability-you-need/"/></url><url><loc>https://llmdeepdive.com/lessons/0-orientation-and-foundations/0.4-vectors-matrices-dot-products/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/0-orientation-and-foundations/0.4-vectors-matrices-dot-products/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/0-orientation-and-foundations/0.4-vectors-matrices-dot-products/"/></url><url><loc>https://llmdeepdive.com/lessons/0-orientation-and-foundations/0.5-matrix-multiplication-transformation/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/0-orientation-and-foundations/0.5-matrix-multiplication-transformation/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/0-orientation-and-foundations/0.5-matrix-multiplication-transformation/"/></url><url><loc>https://llmdeepdive.com/lessons/0-orientation-and-foundations/0.6-derivatives-gradients/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/0-orientation-and-foundations/0.6-derivatives-gradients/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/0-orientation-and-foundations/0.6-derivatives-gradients/"/></url><url><loc>https://llmdeepdive.com/lessons/0-orientation-and-foundations/0.7-python-numpy-pytorch/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/0-orientation-and-foundations/0.7-python-numpy-pytorch/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/0-orientation-and-foundations/0.7-python-numpy-pytorch/"/></url><url><loc>https://llmdeepdive.com/lessons/0-orientation-and-foundations/0.8-how-to-read-ml-paper/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/0-orientation-and-foundations/0.8-how-to-read-ml-paper/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/0-orientation-and-foundations/0.8-how-to-read-ml-paper/"/></url><url><loc>https://llmdeepdive.com/lessons/1-text-to-tensors/1.1-why-computers-cant-read/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/1-text-to-tensors/1.1-why-computers-cant-read/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/1-text-to-tensors/1.1-why-computers-cant-read/"/></url><url><loc>https://llmdeepdive.com/lessons/1-text-to-tensors/1.2-tokenization-i/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/1-text-to-tensors/1.2-tokenization-i/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/1-text-to-tensors/1.2-tokenization-i/"/></url><url><loc>https://llmdeepdive.com/lessons/1-text-to-tensors/1.3-bpe-step-by-step/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/1-text-to-tensors/1.3-bpe-step-by-step/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/1-text-to-tensors/1.3-bpe-step-by-step/"/></url><url><loc>https://llmdeepdive.com/lessons/1-text-to-tensors/1.4-sentencepiece-byte-level-token-counts/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/1-text-to-tensors/1.4-sentencepiece-byte-level-token-counts/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/1-text-to-tensors/1.4-sentencepiece-byte-level-token-counts/"/></url><url><loc>https://llmdeepdive.com/lessons/1-text-to-tensors/1.5-embeddings-meaning-as-geometry/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/1-text-to-tensors/1.5-embeddings-meaning-as-geometry/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/1-text-to-tensors/1.5-embeddings-meaning-as-geometry/"/></url><url><loc>https://llmdeepdive.com/lessons/1-text-to-tensors/1.6-word2vec-glove-analogy/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/1-text-to-tensors/1.6-word2vec-glove-analogy/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/1-text-to-tensors/1.6-word2vec-glove-analogy/"/></url><url><loc>https://llmdeepdive.com/lessons/1-text-to-tensors/1.7-entropy-perplexity/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/1-text-to-tensors/1.7-entropy-perplexity/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/1-text-to-tensors/1.7-entropy-perplexity/"/></url><url><loc>https://llmdeepdive.com/lessons/2-neural-network-fundamentals/2.1-perceptron/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/2-neural-network-fundamentals/2.1-perceptron/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/2-neural-network-fundamentals/2.1-perceptron/"/></url><url><loc>https://llmdeepdive.com/lessons/2-neural-network-fundamentals/2.2-multilayer-perceptrons-nonlinearity/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/2-neural-network-fundamentals/2.2-multilayer-perceptrons-nonlinearity/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/2-neural-network-fundamentals/2.2-multilayer-perceptrons-nonlinearity/"/></url><url><loc>https://llmdeepdive.com/lessons/2-neural-network-fundamentals/2.3-activation-functions/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/2-neural-network-fundamentals/2.3-activation-functions/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/2-neural-network-fundamentals/2.3-activation-functions/"/></url><url><loc>https://llmdeepdive.com/lessons/2-neural-network-fundamentals/2.4-loss-functions-cross-entropy/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/2-neural-network-fundamentals/2.4-loss-functions-cross-entropy/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/2-neural-network-fundamentals/2.4-loss-functions-cross-entropy/"/></url><url><loc>https://llmdeepdive.com/lessons/2-neural-network-fundamentals/2.5-backpropagation/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/2-neural-network-fundamentals/2.5-backpropagation/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/2-neural-network-fundamentals/2.5-backpropagation/"/></url><url><loc>https://llmdeepdive.com/lessons/2-neural-network-fundamentals/2.6-gradient-descent-loss-landscape/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/2-neural-network-fundamentals/2.6-gradient-descent-loss-landscape/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/2-neural-network-fundamentals/2.6-gradient-descent-loss-landscape/"/></url><url><loc>https://llmdeepdive.com/lessons/2-neural-network-fundamentals/2.7-optimizers/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/2-neural-network-fundamentals/2.7-optimizers/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/2-neural-network-fundamentals/2.7-optimizers/"/></url><url><loc>https://llmdeepdive.com/lessons/2-neural-network-fundamentals/2.8-initialization-normalization-residuals/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/2-neural-network-fundamentals/2.8-initialization-normalization-residuals/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/2-neural-network-fundamentals/2.8-initialization-normalization-residuals/"/></url><url><loc>https://llmdeepdive.com/lessons/2-neural-network-fundamentals/2.9-overfitting-regularization-bitter-lesson/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/2-neural-network-fundamentals/2.9-overfitting-regularization-bitter-lesson/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/2-neural-network-fundamentals/2.9-overfitting-regularization-bitter-lesson/"/></url><url><loc>https://llmdeepdive.com/lessons/3-sequence-models/3.1-modelling-sequences/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/3-sequence-models/3.1-modelling-sequences/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/3-sequence-models/3.1-modelling-sequences/"/></url><url><loc>https://llmdeepdive.com/lessons/3-sequence-models/3.2-rnns-vanishing-gradient/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/3-sequence-models/3.2-rnns-vanishing-gradient/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/3-sequence-models/3.2-rnns-vanishing-gradient/"/></url><url><loc>https://llmdeepdive.com/lessons/3-sequence-models/3.3-lstm-gru-gates/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/3-sequence-models/3.3-lstm-gru-gates/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/3-sequence-models/3.3-lstm-gru-gates/"/></url><url><loc>https://llmdeepdive.com/lessons/3-sequence-models/3.4-seq2seq-encoder-decoder-bottleneck/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/3-sequence-models/3.4-seq2seq-encoder-decoder-bottleneck/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/3-sequence-models/3.4-seq2seq-encoder-decoder-bottleneck/"/></url><url><loc>https://llmdeepdive.com/lessons/3-sequence-models/3.5-bahdanau-attention/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/3-sequence-models/3.5-bahdanau-attention/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/3-sequence-models/3.5-bahdanau-attention/"/></url><url><loc>https://llmdeepdive.com/lessons/4-transformer/4.1-attention-is-all-you-need-in-context/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/4-transformer/4.1-attention-is-all-you-need-in-context/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/4-transformer/4.1-attention-is-all-you-need-in-context/"/></url><url><loc>https://llmdeepdive.com/lessons/4-transformer/4.2-self-attention/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/4-transformer/4.2-self-attention/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/4-transformer/4.2-self-attention/"/></url><url><loc>https://llmdeepdive.com/lessons/4-transformer/4.3-scaled-dot-product-attention/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/4-transformer/4.3-scaled-dot-product-attention/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/4-transformer/4.3-scaled-dot-product-attention/"/></url><url><loc>https://llmdeepdive.com/lessons/4-transformer/4.4-causal-masking/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/4-transformer/4.4-causal-masking/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/4-transformer/4.4-causal-masking/"/></url><url><loc>https://llmdeepdive.com/lessons/4-transformer/4.5-multi-head-attention/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/4-transformer/4.5-multi-head-attention/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/4-transformer/4.5-multi-head-attention/"/></url><url><loc>https://llmdeepdive.com/lessons/4-transformer/4.6-positional-encoding-sinusoidal-learned/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/4-transformer/4.6-positional-encoding-sinusoidal-learned/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/4-transformer/4.6-positional-encoding-sinusoidal-learned/"/></url><url><loc>https://llmdeepdive.com/lessons/4-transformer/4.7-rope/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/4-transformer/4.7-rope/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/4-transformer/4.7-rope/"/></url><url><loc>https://llmdeepdive.com/lessons/4-transformer/4.8-alibi-relative-bias/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/4-transformer/4.8-alibi-relative-bias/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/4-transformer/4.8-alibi-relative-bias/"/></url><url><loc>https://llmdeepdive.com/lessons/4-transformer/4.9-feed-forward-block/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/4-transformer/4.9-feed-forward-block/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/4-transformer/4.9-feed-forward-block/"/></url><url><loc>https://llmdeepdive.com/lessons/4-transformer/4.10-residuals-pre-norm-post-norm/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/4-transformer/4.10-residuals-pre-norm-post-norm/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/4-transformer/4.10-residuals-pre-norm-post-norm/"/></url><url><loc>https://llmdeepdive.com/lessons/4-transformer/4.11-full-transformer-block/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/4-transformer/4.11-full-transformer-block/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/4-transformer/4.11-full-transformer-block/"/></url><url><loc>https://llmdeepdive.com/lessons/4-transformer/4.12-transformer-families/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/4-transformer/4.12-transformer-families/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/4-transformer/4.12-transformer-families/"/></url><url><loc>https://llmdeepdive.com/lessons/4-transformer/4.13-build-gpt-from-scratch/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/4-transformer/4.13-build-gpt-from-scratch/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/4-transformer/4.13-build-gpt-from-scratch/"/></url><url><loc>https://llmdeepdive.com/lessons/4-transformer/4.14-reading-real-weights/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/4-transformer/4.14-reading-real-weights/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/4-transformer/4.14-reading-real-weights/"/></url><url><loc>https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.1-pretraining-objectives/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.1-pretraining-objectives/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.1-pretraining-objectives/"/></url><url><loc>https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.2-data-pipeline/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.2-data-pipeline/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.2-data-pipeline/"/></url><url><loc>https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.3-training-tokenizer/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.3-training-tokenizer/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.3-training-tokenizer/"/></url><url><loc>https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.4-kaplan-scaling-laws/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.4-kaplan-scaling-laws/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.4-kaplan-scaling-laws/"/></url><url><loc>https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.5-chinchilla-compute-optimality/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.5-chinchilla-compute-optimality/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.5-chinchilla-compute-optimality/"/></url><url><loc>https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.6-inference-aware-scaling/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.6-inference-aware-scaling/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.6-inference-aware-scaling/"/></url><url><loc>https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.7-data-parallelism-zero-fsdp/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.7-data-parallelism-zero-fsdp/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.7-data-parallelism-zero-fsdp/"/></url><url><loc>https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.8-model-parallelism/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.8-model-parallelism/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.8-model-parallelism/"/></url><url><loc>https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.9-mixed-precision/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.9-mixed-precision/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.9-mixed-precision/"/></url><url><loc>https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.10-gradient-checkpointing/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.10-gradient-checkpointing/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.10-gradient-checkpointing/"/></url><url><loc>https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.11-learning-rate-schedules/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.11-learning-rate-schedules/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.11-learning-rate-schedules/"/></url><url><loc>https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.12-training-instabilities/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.12-training-instabilities/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.12-training-instabilities/"/></url><url><loc>https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.13-pretraining-costs/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.13-pretraining-costs/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.13-pretraining-costs/"/></url><url><loc>https://llmdeepdive.com/lessons/6-post-training-and-alignment/6.1-why-base-models-arent-assistants/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/6-post-training-and-alignment/6.1-why-base-models-arent-assistants/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/6-post-training-and-alignment/6.1-why-base-models-arent-assistants/"/></url><url><loc>https://llmdeepdive.com/lessons/6-post-training-and-alignment/6.2-supervised-fine-tuning-and-instruction-data/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/6-post-training-and-alignment/6.2-supervised-fine-tuning-and-instruction-data/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/6-post-training-and-alignment/6.2-supervised-fine-tuning-and-instruction-data/"/></url><url><loc>https://llmdeepdive.com/lessons/6-post-training-and-alignment/6.3-lora-qlora-and-peft/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/6-post-training-and-alignment/6.3-lora-qlora-and-peft/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/6-post-training-and-alignment/6.3-lora-qlora-and-peft/"/></url><url><loc>https://llmdeepdive.com/lessons/6-post-training-and-alignment/6.4-reward-models-and-human-preference-data/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/6-post-training-and-alignment/6.4-reward-models-and-human-preference-data/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/6-post-training-and-alignment/6.4-reward-models-and-human-preference-data/"/></url><url><loc>https://llmdeepdive.com/lessons/6-post-training-and-alignment/6.5-rlhf-with-ppo/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/6-post-training-and-alignment/6.5-rlhf-with-ppo/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/6-post-training-and-alignment/6.5-rlhf-with-ppo/"/></url><url><loc>https://llmdeepdive.com/lessons/6-post-training-and-alignment/6.6-dpo-skipping-the-reward-model/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/6-post-training-and-alignment/6.6-dpo-skipping-the-reward-model/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/6-post-training-and-alignment/6.6-dpo-skipping-the-reward-model/"/></url><url><loc>https://llmdeepdive.com/lessons/6-post-training-and-alignment/6.7-orpo-kto-simpo-and-the-alignment-zoo/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/6-post-training-and-alignment/6.7-orpo-kto-simpo-and-the-alignment-zoo/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/6-post-training-and-alignment/6.7-orpo-kto-simpo-and-the-alignment-zoo/"/></url><url><loc>https://llmdeepdive.com/lessons/6-post-training-and-alignment/6.8-grpo-and-rl-on-verifiable-rewards/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/6-post-training-and-alignment/6.8-grpo-and-rl-on-verifiable-rewards/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/6-post-training-and-alignment/6.8-grpo-and-rl-on-verifiable-rewards/"/></url><url><loc>https://llmdeepdive.com/lessons/6-post-training-and-alignment/6.9-reasoning-models-test-time-compute-and-long-cot/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/6-post-training-and-alignment/6.9-reasoning-models-test-time-compute-and-long-cot/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/6-post-training-and-alignment/6.9-reasoning-models-test-time-compute-and-long-cot/"/></url><url><loc>https://llmdeepdive.com/lessons/6-post-training-and-alignment/6.10-constitutional-ai-and-rlaif/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/6-post-training-and-alignment/6.10-constitutional-ai-and-rlaif/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/6-post-training-and-alignment/6.10-constitutional-ai-and-rlaif/"/></url><url><loc>https://llmdeepdive.com/lessons/6-post-training-and-alignment/6.11-distillation-making-small-models-punch-up/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/6-post-training-and-alignment/6.11-distillation-making-small-models-punch-up/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/6-post-training-and-alignment/6.11-distillation-making-small-models-punch-up/"/></url><url><loc>https://llmdeepdive.com/lessons/7-inference-and-efficiency/7.1-what-actually-happens-when-you-hit-send/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/7-inference-and-efficiency/7.1-what-actually-happens-when-you-hit-send/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/7-inference-and-efficiency/7.1-what-actually-happens-when-you-hit-send/"/></url><url><loc>https://llmdeepdive.com/lessons/7-inference-and-efficiency/7.2-the-kv-cache/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/7-inference-and-efficiency/7.2-the-kv-cache/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/7-inference-and-efficiency/7.2-the-kv-cache/"/></url><url><loc>https://llmdeepdive.com/lessons/7-inference-and-efficiency/7.3-prefill-vs-decode-two-different-machines/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/7-inference-and-efficiency/7.3-prefill-vs-decode-two-different-machines/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/7-inference-and-efficiency/7.3-prefill-vs-decode-two-different-machines/"/></url><url><loc>https://llmdeepdive.com/lessons/7-inference-and-efficiency/7.4-sampling/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/7-inference-and-efficiency/7.4-sampling/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/7-inference-and-efficiency/7.4-sampling/"/></url><url><loc>https://llmdeepdive.com/lessons/7-inference-and-efficiency/7.5-beam-search-speculative-decoding-and-medusa/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/7-inference-and-efficiency/7.5-beam-search-speculative-decoding-and-medusa/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/7-inference-and-efficiency/7.5-beam-search-speculative-decoding-and-medusa/"/></url><url><loc>https://llmdeepdive.com/lessons/7-inference-and-efficiency/7.6-mqa-gqa-and-mla/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/7-inference-and-efficiency/7.6-mqa-gqa-and-mla/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/7-inference-and-efficiency/7.6-mqa-gqa-and-mla/"/></url><url><loc>https://llmdeepdive.com/lessons/7-inference-and-efficiency/7.7-flashattention-and-io-awareness/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/7-inference-and-efficiency/7.7-flashattention-and-io-awareness/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/7-inference-and-efficiency/7.7-flashattention-and-io-awareness/"/></url><url><loc>https://llmdeepdive.com/lessons/7-inference-and-efficiency/7.8-pagedattention-and-continuous-batching/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/7-inference-and-efficiency/7.8-pagedattention-and-continuous-batching/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/7-inference-and-efficiency/7.8-pagedattention-and-continuous-batching/"/></url><url><loc>https://llmdeepdive.com/lessons/7-inference-and-efficiency/7.9-quantization-i-int8-int4-the-basics/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/7-inference-and-efficiency/7.9-quantization-i-int8-int4-the-basics/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/7-inference-and-efficiency/7.9-quantization-i-int8-int4-the-basics/"/></url><url><loc>https://llmdeepdive.com/lessons/7-inference-and-efficiency/7.10-quantization-ii-gptq-awq-gguf-qat/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/7-inference-and-efficiency/7.10-quantization-ii-gptq-awq-gguf-qat/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/7-inference-and-efficiency/7.10-quantization-ii-gptq-awq-gguf-qat/"/></url><url><loc>https://llmdeepdive.com/lessons/7-inference-and-efficiency/7.11-serving-stacks-vllm-sglang-tensorrt-llm/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/7-inference-and-efficiency/7.11-serving-stacks-vllm-sglang-tensorrt-llm/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/7-inference-and-efficiency/7.11-serving-stacks-vllm-sglang-tensorrt-llm/"/></url><url><loc>https://llmdeepdive.com/lessons/7-inference-and-efficiency/7.12-running-models-locally/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/7-inference-and-efficiency/7.12-running-models-locally/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/7-inference-and-efficiency/7.12-running-models-locally/"/></url><url><loc>https://llmdeepdive.com/pt-br/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/"/></url><url><loc>https://llmdeepdive.com/pt-br/explore/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/explore/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/explore/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/0-orientation-and-foundations/0.1-what-is-a-language-model/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/0-orientation-and-foundations/0.1-what-is-a-language-model/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/0-orientation-and-foundations/0.1-what-is-a-language-model/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/0-orientation-and-foundations/0.2-history-ngram-to-gpt/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/0-orientation-and-foundations/0.2-history-ngram-to-gpt/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/0-orientation-and-foundations/0.2-history-ngram-to-gpt/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/0-orientation-and-foundations/0.3-probability-you-need/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/0-orientation-and-foundations/0.3-probability-you-need/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/0-orientation-and-foundations/0.3-probability-you-need/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/0-orientation-and-foundations/0.4-vectors-matrices-dot-products/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/0-orientation-and-foundations/0.4-vectors-matrices-dot-products/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/0-orientation-and-foundations/0.4-vectors-matrices-dot-products/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/0-orientation-and-foundations/0.5-matrix-multiplication-transformation/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/0-orientation-and-foundations/0.5-matrix-multiplication-transformation/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/0-orientation-and-foundations/0.5-matrix-multiplication-transformation/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/0-orientation-and-foundations/0.6-derivatives-gradients/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/0-orientation-and-foundations/0.6-derivatives-gradients/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/0-orientation-and-foundations/0.6-derivatives-gradients/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/0-orientation-and-foundations/0.7-python-numpy-pytorch/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/0-orientation-and-foundations/0.7-python-numpy-pytorch/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/0-orientation-and-foundations/0.7-python-numpy-pytorch/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/0-orientation-and-foundations/0.8-how-to-read-ml-paper/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/0-orientation-and-foundations/0.8-how-to-read-ml-paper/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/0-orientation-and-foundations/0.8-how-to-read-ml-paper/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/1-text-to-tensors/1.1-why-computers-cant-read/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/1-text-to-tensors/1.1-why-computers-cant-read/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/1-text-to-tensors/1.1-why-computers-cant-read/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/1-text-to-tensors/1.2-tokenization-i/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/1-text-to-tensors/1.2-tokenization-i/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/1-text-to-tensors/1.2-tokenization-i/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/1-text-to-tensors/1.3-bpe-step-by-step/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/1-text-to-tensors/1.3-bpe-step-by-step/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/1-text-to-tensors/1.3-bpe-step-by-step/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/1-text-to-tensors/1.4-sentencepiece-byte-level-token-counts/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/1-text-to-tensors/1.4-sentencepiece-byte-level-token-counts/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/1-text-to-tensors/1.4-sentencepiece-byte-level-token-counts/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/1-text-to-tensors/1.5-embeddings-meaning-as-geometry/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/1-text-to-tensors/1.5-embeddings-meaning-as-geometry/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/1-text-to-tensors/1.5-embeddings-meaning-as-geometry/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/1-text-to-tensors/1.6-word2vec-glove-analogy/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/1-text-to-tensors/1.6-word2vec-glove-analogy/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/1-text-to-tensors/1.6-word2vec-glove-analogy/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/1-text-to-tensors/1.7-entropy-perplexity/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/1-text-to-tensors/1.7-entropy-perplexity/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/1-text-to-tensors/1.7-entropy-perplexity/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/2-neural-network-fundamentals/2.1-perceptron/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/2-neural-network-fundamentals/2.1-perceptron/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/2-neural-network-fundamentals/2.1-perceptron/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/2-neural-network-fundamentals/2.2-multilayer-perceptrons-nonlinearity/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/2-neural-network-fundamentals/2.2-multilayer-perceptrons-nonlinearity/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/2-neural-network-fundamentals/2.2-multilayer-perceptrons-nonlinearity/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/2-neural-network-fundamentals/2.3-activation-functions/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/2-neural-network-fundamentals/2.3-activation-functions/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/2-neural-network-fundamentals/2.3-activation-functions/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/2-neural-network-fundamentals/2.4-loss-functions-cross-entropy/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/2-neural-network-fundamentals/2.4-loss-functions-cross-entropy/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/2-neural-network-fundamentals/2.4-loss-functions-cross-entropy/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/2-neural-network-fundamentals/2.5-backpropagation/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/2-neural-network-fundamentals/2.5-backpropagation/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/2-neural-network-fundamentals/2.5-backpropagation/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/2-neural-network-fundamentals/2.6-gradient-descent-loss-landscape/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/2-neural-network-fundamentals/2.6-gradient-descent-loss-landscape/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/2-neural-network-fundamentals/2.6-gradient-descent-loss-landscape/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/2-neural-network-fundamentals/2.7-optimizers/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/2-neural-network-fundamentals/2.7-optimizers/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/2-neural-network-fundamentals/2.7-optimizers/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/2-neural-network-fundamentals/2.8-initialization-normalization-residuals/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/2-neural-network-fundamentals/2.8-initialization-normalization-residuals/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/2-neural-network-fundamentals/2.8-initialization-normalization-residuals/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/2-neural-network-fundamentals/2.9-overfitting-regularization-bitter-lesson/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/2-neural-network-fundamentals/2.9-overfitting-regularization-bitter-lesson/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/2-neural-network-fundamentals/2.9-overfitting-regularization-bitter-lesson/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/3-sequence-models/3.1-modelling-sequences/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/3-sequence-models/3.1-modelling-sequences/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/3-sequence-models/3.1-modelling-sequences/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/3-sequence-models/3.2-rnns-vanishing-gradient/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/3-sequence-models/3.2-rnns-vanishing-gradient/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/3-sequence-models/3.2-rnns-vanishing-gradient/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/3-sequence-models/3.3-lstm-gru-gates/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/3-sequence-models/3.3-lstm-gru-gates/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/3-sequence-models/3.3-lstm-gru-gates/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/3-sequence-models/3.4-seq2seq-encoder-decoder-bottleneck/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/3-sequence-models/3.4-seq2seq-encoder-decoder-bottleneck/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/3-sequence-models/3.4-seq2seq-encoder-decoder-bottleneck/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/3-sequence-models/3.5-bahdanau-attention/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/3-sequence-models/3.5-bahdanau-attention/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/3-sequence-models/3.5-bahdanau-attention/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/4-transformer/4.1-attention-is-all-you-need-in-context/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/4-transformer/4.1-attention-is-all-you-need-in-context/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/4-transformer/4.1-attention-is-all-you-need-in-context/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/4-transformer/4.2-self-attention/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/4-transformer/4.2-self-attention/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/4-transformer/4.2-self-attention/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/4-transformer/4.3-scaled-dot-product-attention/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/4-transformer/4.3-scaled-dot-product-attention/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/4-transformer/4.3-scaled-dot-product-attention/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/4-transformer/4.4-causal-masking/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/4-transformer/4.4-causal-masking/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/4-transformer/4.4-causal-masking/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/4-transformer/4.5-multi-head-attention/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/4-transformer/4.5-multi-head-attention/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/4-transformer/4.5-multi-head-attention/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/4-transformer/4.6-positional-encoding-sinusoidal-learned/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/4-transformer/4.6-positional-encoding-sinusoidal-learned/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/4-transformer/4.6-positional-encoding-sinusoidal-learned/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/4-transformer/4.7-rope/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/4-transformer/4.7-rope/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/4-transformer/4.7-rope/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/4-transformer/4.8-alibi-relative-bias/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/4-transformer/4.8-alibi-relative-bias/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/4-transformer/4.8-alibi-relative-bias/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/4-transformer/4.9-feed-forward-block/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/4-transformer/4.9-feed-forward-block/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/4-transformer/4.9-feed-forward-block/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/4-transformer/4.10-residuals-pre-norm-post-norm/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/4-transformer/4.10-residuals-pre-norm-post-norm/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/4-transformer/4.10-residuals-pre-norm-post-norm/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/4-transformer/4.11-full-transformer-block/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/4-transformer/4.11-full-transformer-block/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/4-transformer/4.11-full-transformer-block/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/4-transformer/4.12-transformer-families/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/4-transformer/4.12-transformer-families/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/4-transformer/4.12-transformer-families/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/4-transformer/4.13-build-gpt-from-scratch/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/4-transformer/4.13-build-gpt-from-scratch/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/4-transformer/4.13-build-gpt-from-scratch/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/4-transformer/4.14-reading-real-weights/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/4-transformer/4.14-reading-real-weights/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/4-transformer/4.14-reading-real-weights/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.1-pretraining-objectives/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.1-pretraining-objectives/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.1-pretraining-objectives/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.2-data-pipeline/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.2-data-pipeline/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.2-data-pipeline/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.3-training-tokenizer/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.3-training-tokenizer/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.3-training-tokenizer/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.4-kaplan-scaling-laws/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.4-kaplan-scaling-laws/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.4-kaplan-scaling-laws/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.5-chinchilla-compute-optimality/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.5-chinchilla-compute-optimality/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.5-chinchilla-compute-optimality/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.6-inference-aware-scaling/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.6-inference-aware-scaling/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.6-inference-aware-scaling/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.7-data-parallelism-zero-fsdp/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.7-data-parallelism-zero-fsdp/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.7-data-parallelism-zero-fsdp/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.8-model-parallelism/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.8-model-parallelism/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.8-model-parallelism/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.9-mixed-precision/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.9-mixed-precision/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.9-mixed-precision/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.10-gradient-checkpointing/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.10-gradient-checkpointing/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.10-gradient-checkpointing/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.11-learning-rate-schedules/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.11-learning-rate-schedules/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.11-learning-rate-schedules/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.12-training-instabilities/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.12-training-instabilities/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.12-training-instabilities/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.13-pretraining-costs/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/5-pretraining-at-scale/5.13-pretraining-costs/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/5-pretraining-at-scale/5.13-pretraining-costs/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/6-post-training-and-alignment/6.1-why-base-models-arent-assistants/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/6-post-training-and-alignment/6.1-why-base-models-arent-assistants/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/6-post-training-and-alignment/6.1-why-base-models-arent-assistants/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/6-post-training-and-alignment/6.2-supervised-fine-tuning-and-instruction-data/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/6-post-training-and-alignment/6.2-supervised-fine-tuning-and-instruction-data/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/6-post-training-and-alignment/6.2-supervised-fine-tuning-and-instruction-data/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/6-post-training-and-alignment/6.3-lora-qlora-and-peft/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/6-post-training-and-alignment/6.3-lora-qlora-and-peft/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/6-post-training-and-alignment/6.3-lora-qlora-and-peft/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/6-post-training-and-alignment/6.4-reward-models-and-human-preference-data/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/6-post-training-and-alignment/6.4-reward-models-and-human-preference-data/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/6-post-training-and-alignment/6.4-reward-models-and-human-preference-data/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/6-post-training-and-alignment/6.5-rlhf-with-ppo/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/6-post-training-and-alignment/6.5-rlhf-with-ppo/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/6-post-training-and-alignment/6.5-rlhf-with-ppo/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/6-post-training-and-alignment/6.6-dpo-skipping-the-reward-model/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/6-post-training-and-alignment/6.6-dpo-skipping-the-reward-model/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/6-post-training-and-alignment/6.6-dpo-skipping-the-reward-model/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/6-post-training-and-alignment/6.7-orpo-kto-simpo-and-the-alignment-zoo/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/6-post-training-and-alignment/6.7-orpo-kto-simpo-and-the-alignment-zoo/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/6-post-training-and-alignment/6.7-orpo-kto-simpo-and-the-alignment-zoo/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/6-post-training-and-alignment/6.8-grpo-and-rl-on-verifiable-rewards/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/6-post-training-and-alignment/6.8-grpo-and-rl-on-verifiable-rewards/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/6-post-training-and-alignment/6.8-grpo-and-rl-on-verifiable-rewards/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/6-post-training-and-alignment/6.9-reasoning-models-test-time-compute-and-long-cot/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/6-post-training-and-alignment/6.9-reasoning-models-test-time-compute-and-long-cot/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/6-post-training-and-alignment/6.9-reasoning-models-test-time-compute-and-long-cot/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/6-post-training-and-alignment/6.10-constitutional-ai-and-rlaif/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/6-post-training-and-alignment/6.10-constitutional-ai-and-rlaif/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/6-post-training-and-alignment/6.10-constitutional-ai-and-rlaif/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/6-post-training-and-alignment/6.11-distillation-making-small-models-punch-up/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/6-post-training-and-alignment/6.11-distillation-making-small-models-punch-up/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/6-post-training-and-alignment/6.11-distillation-making-small-models-punch-up/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/7-inference-and-efficiency/7.1-what-actually-happens-when-you-hit-send/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/7-inference-and-efficiency/7.1-what-actually-happens-when-you-hit-send/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/7-inference-and-efficiency/7.1-what-actually-happens-when-you-hit-send/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/7-inference-and-efficiency/7.2-the-kv-cache/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/7-inference-and-efficiency/7.2-the-kv-cache/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/7-inference-and-efficiency/7.2-the-kv-cache/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/7-inference-and-efficiency/7.3-prefill-vs-decode-two-different-machines/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/7-inference-and-efficiency/7.3-prefill-vs-decode-two-different-machines/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/7-inference-and-efficiency/7.3-prefill-vs-decode-two-different-machines/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/7-inference-and-efficiency/7.4-sampling/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/7-inference-and-efficiency/7.4-sampling/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/7-inference-and-efficiency/7.4-sampling/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/7-inference-and-efficiency/7.5-beam-search-speculative-decoding-and-medusa/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/7-inference-and-efficiency/7.5-beam-search-speculative-decoding-and-medusa/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/7-inference-and-efficiency/7.5-beam-search-speculative-decoding-and-medusa/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/7-inference-and-efficiency/7.6-mqa-gqa-and-mla/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/7-inference-and-efficiency/7.6-mqa-gqa-and-mla/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/7-inference-and-efficiency/7.6-mqa-gqa-and-mla/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/7-inference-and-efficiency/7.7-flashattention-and-io-awareness/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/7-inference-and-efficiency/7.7-flashattention-and-io-awareness/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/7-inference-and-efficiency/7.7-flashattention-and-io-awareness/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/7-inference-and-efficiency/7.8-pagedattention-and-continuous-batching/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/7-inference-and-efficiency/7.8-pagedattention-and-continuous-batching/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/7-inference-and-efficiency/7.8-pagedattention-and-continuous-batching/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/7-inference-and-efficiency/7.9-quantization-i-int8-int4-the-basics/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/7-inference-and-efficiency/7.9-quantization-i-int8-int4-the-basics/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/7-inference-and-efficiency/7.9-quantization-i-int8-int4-the-basics/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/7-inference-and-efficiency/7.10-quantization-ii-gptq-awq-gguf-qat/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/7-inference-and-efficiency/7.10-quantization-ii-gptq-awq-gguf-qat/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/7-inference-and-efficiency/7.10-quantization-ii-gptq-awq-gguf-qat/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/7-inference-and-efficiency/7.11-serving-stacks-vllm-sglang-tensorrt-llm/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/7-inference-and-efficiency/7.11-serving-stacks-vllm-sglang-tensorrt-llm/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/7-inference-and-efficiency/7.11-serving-stacks-vllm-sglang-tensorrt-llm/"/></url><url><loc>https://llmdeepdive.com/pt-br/lessons/7-inference-and-efficiency/7.12-running-models-locally/</loc><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/lessons/7-inference-and-efficiency/7.12-running-models-locally/"/><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/lessons/7-inference-and-efficiency/7.12-running-models-locally/"/></url><url><loc>https://llmdeepdive.com/pt-br/tracks/</loc><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/tracks/"/><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/tracks/"/></url><url><loc>https://llmdeepdive.com/pt-br/tracks/0-orientation-and-foundations/</loc><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/tracks/0-orientation-and-foundations/"/><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/tracks/0-orientation-and-foundations/"/></url><url><loc>https://llmdeepdive.com/pt-br/tracks/1-text-to-tensors/</loc><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/tracks/1-text-to-tensors/"/><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/tracks/1-text-to-tensors/"/></url><url><loc>https://llmdeepdive.com/pt-br/tracks/2-neural-network-fundamentals/</loc><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/tracks/2-neural-network-fundamentals/"/><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/tracks/2-neural-network-fundamentals/"/></url><url><loc>https://llmdeepdive.com/pt-br/tracks/3-sequence-models/</loc><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/tracks/3-sequence-models/"/><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/tracks/3-sequence-models/"/></url><url><loc>https://llmdeepdive.com/pt-br/tracks/4-transformer/</loc><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/tracks/4-transformer/"/><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/tracks/4-transformer/"/></url><url><loc>https://llmdeepdive.com/pt-br/tracks/5-pretraining-at-scale/</loc><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/tracks/5-pretraining-at-scale/"/><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/tracks/5-pretraining-at-scale/"/></url><url><loc>https://llmdeepdive.com/pt-br/tracks/6-post-training-and-alignment/</loc><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/tracks/6-post-training-and-alignment/"/><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/tracks/6-post-training-and-alignment/"/></url><url><loc>https://llmdeepdive.com/pt-br/tracks/7-inference-and-efficiency/</loc><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/tracks/7-inference-and-efficiency/"/><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/tracks/7-inference-and-efficiency/"/></url><url><loc>https://llmdeepdive.com/tracks/</loc><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/tracks/"/><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/tracks/"/></url><url><loc>https://llmdeepdive.com/tracks/0-orientation-and-foundations/</loc><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/tracks/0-orientation-and-foundations/"/><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/tracks/0-orientation-and-foundations/"/></url><url><loc>https://llmdeepdive.com/tracks/1-text-to-tensors/</loc><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/tracks/1-text-to-tensors/"/><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/tracks/1-text-to-tensors/"/></url><url><loc>https://llmdeepdive.com/tracks/2-neural-network-fundamentals/</loc><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/tracks/2-neural-network-fundamentals/"/><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/tracks/2-neural-network-fundamentals/"/></url><url><loc>https://llmdeepdive.com/tracks/3-sequence-models/</loc><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/tracks/3-sequence-models/"/><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/tracks/3-sequence-models/"/></url><url><loc>https://llmdeepdive.com/tracks/4-transformer/</loc><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/tracks/4-transformer/"/><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/tracks/4-transformer/"/></url><url><loc>https://llmdeepdive.com/tracks/5-pretraining-at-scale/</loc><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/tracks/5-pretraining-at-scale/"/><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/tracks/5-pretraining-at-scale/"/></url><url><loc>https://llmdeepdive.com/tracks/6-post-training-and-alignment/</loc><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/tracks/6-post-training-and-alignment/"/><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/tracks/6-post-training-and-alignment/"/></url><url><loc>https://llmdeepdive.com/tracks/7-inference-and-efficiency/</loc><xhtml:link rel="alternate" hreflang="pt-BR" href="https://llmdeepdive.com/pt-br/tracks/7-inference-and-efficiency/"/><xhtml:link rel="alternate" hreflang="en" href="https://llmdeepdive.com/tracks/7-inference-and-efficiency/"/></url></urlset>