\documentclass[11pt,a4paper]{article} % ===== Packages ===== \usepackage[utf8]{inputenc} \usepackage[T1]{fontenc} \usepackage{newtxtext,newtxmath} \usepackage[margin=1in]{geometry} \usepackage{amsmath} \usepackage{booktabs} \usepackage{graphicx} \usepackage{hyperref} \usepackage{float} \usepackage{caption} \usepackage{enumitem} \usepackage{url} \usepackage{microtype} \DeclareMathOperator{\Var}{Var} \title{End-to-End Training of a 1.2B Transformer with AstrAI \\ \large Data Pipeline, Distributed Training, and Ablations on Optimizer, Initialization, and BF16 Numerical Stability} \author{AstrAI Contributors} \date{} \begin{document} \maketitle \begin{abstract} We present {\sc AstrAI}, an open-source framework for end-to-end training of a 1.2B-parameter Transformer on $\sim$25B tokens. The pipeline covers JSON-driven BBPE preprocessing with multi-strategy packing, tiered storage backends, and a companion SFT pipeline ({\sc Alembic}) with MinHash deduplication. The 24-layer GQA-SwiGLU decoder is trained with a hybrid Muon/AdamW optimizer and WSD scheduling under DDP/FSDP. Supervised fine-tuning on deduplicated bilingual instructions reduces loss from $\sim$2.1 to $\sim$1.5 over $\sim$3{,}800~steps; DPO alignment ($\beta=0.1$, cosine schedule) on model-generated preference pairs shows stable training-loss convergence without over-optimisation. A BF16 stability analysis shows that GPT-2 residual scaling ($\sigma_0 = 0.02/\sqrt{2L}$) reduces per-block activation variance by a factor of 48, and post-training weight analysis across three checkpoints confirms that residual-scaled projections remain consistently narrower than non-scaled weights across optimizers, initializations, and training budgets. Optimizer ablations demonstrate that the hybrid Muon/AdamW outperforms pure AdamW, with 2D weight matrices benefiting from Muon's orthogonalisation and 1D parameters from AdamW's second-moment adaptation. An SVD effective-rank analysis reveals near-capacity weight utilization, with Q/O projections showing consistently lower effective rank than K/V projections under grouped query attention. \end{abstract} % ====================================================================== \section{Introduction} % ====================================================================== Training a billion-parameter language model end-to-end involves far more than model architecture. Data must be preprocessed and stored efficiently, the training loop must handle distributed parallelism, gradient accumulation, checkpointing, and logging---and numerical pitfalls must be diagnosed and fixed. This paper describes the complete workflow using {\sc AstrAI}~\cite{astrai}, an open-source framework for Transformer training and inference, from JSONL ingestion through pretraining, supervised fine-tuning (SFT) on deduplicated bilingual instructions, and direct preference optimization (DPO) alignment. We also conduct systematic ablations on optimizers (hybrid Muon/AdamW versus pure AdamW) and initializations (GPT-2 residual scaling, Kaiming, and Normal), and analyse a BF16 precision issue encountered along the way. % ====================================================================== \section{Data Pipeline} % ====================================================================== \subsection{Preprocessing} Raw data arrives as JSONL files. The preprocessing pipeline is configured via a JSON specification that defines: \begin{itemize}[nosep] \item \textbf{Tokenization}: BBPE tokenizer (100K vocabulary) with standard special tokens. \item \textbf{Masking}: Declarative loss mask assignment per section (e.g.,~mask user input, compute loss on assistant response). \item \textbf{Packing}: Documents concatenated via \texttt{simple} (sequential), \texttt{bfd} (best-fit decreasing), or \texttt{bfd\_\allowbreak{}split} strategies. \item \textbf{Position IDs}: \texttt{none}, \texttt{doc\_reset} (per-document boundary), or \texttt{continuous}. \item \textbf{Output}: Tokenized sequences written to \texttt{.h5} or \texttt{.bin} shards, auto-split at 100M tokens per shard. \end{itemize} In text-mode sections, individual fields shorter than 50~chars or longer than 2M~chars are skipped during tokenization. \subsection{Storage Backends} Three storage backends serve the DataLoader, trading off memory footprint against access speed for datasets ranging from fine-tuning scale to TB-level pretraining: \begin{itemize}[nosep] \item \textbf{H5Store}: HDF5-based, fully loaded into RAM at init with \texttt{share\_memory\_()} for cross-worker sharing. Fastest random access; requires the dataset to fit in memory. \item \textbf{MmapStore}: Zero-copy memory-mapped \texttt{.bin} files via \texttt{np.memmap}. Data stays on disk, managed by the OS page cache; multiple workers share physical pages without duplication. Suited for large pretraining corpora that exceed RAM. \item \textbf{JsonlStore}: Reads raw \texttt{.jsonl} directly. Lazy mode (\texttt{processor=fn}) keeps only raw text records in memory and defers per-sample tokenization to \texttt{fetch\_record()}, avoiding a pre-tokenized copy entirely---used by DPO/GRPO for on-the-fly training from source files. \end{itemize} A resumable distributed sampler provides seed-based shuffle with epoch/iteration resume. \subsection{SFT Data Cleaning} For supervised fine-tuning (SFT), raw data requires additional curation beyond pretraining tokenization. {\sc Alembic}~\cite{alembic} is a companion pipeline that handles SFT data generation, cleaning, and quality scoring: three-generation strategies (topic-driven, seed-driven, self-instruct), built-in cleaning (HTML/URL/markdown removal, char/word repetition filters), and a MinHash-based near-duplicate detection system~\cite{broder1997syntactic}. Given a set of $P$ hash functions ($P=128$) and a text $T$, the MinHash pipeline proceeds as follows: \begin{enumerate}[nosep,leftmargin=*] \item \textbf{Tokenization}: $T$ is split into character $n$-grams ($n=3$): \begin{equation} \Gamma(T) = \{\,c_i c_{i+1} c_{i+2} \mid i = 1,\dots,|T|-2 \,\}. \end{equation} \item \textbf{Signature}: For each hash function $h_k$, the minimum hash value over all $n$-grams forms the $k$-th element of the fingerprint: \begin{equation} s_k = \min_{t \in \Gamma(T)} h_k(t), \qquad h_k(t) = \operatorname{SHA256}(42 : k : t)_{[0:63]}. \end{equation} The full fingerprint is $\mathbf{s} = (s_1,\dots,s_P)$. \item \textbf{Similarity}: The Jaccard similarity between two sets is estimated by the fraction of agreeing fingerprint positions: \begin{equation} \widehat{J}(\mathbf{s}^{(a)},\mathbf{s}^{(b)}) = \frac{|\{\,k \mid s^{(a)}_k = s^{(b)}_k \,\}|}{P}. \end{equation} \item \textbf{Filtering}: Samples are processed sequentially; a sample is dropped if $\widehat{J}(\mathbf{s}, \mathbf{s}') \ge 0.7$ for any previously kept sample $\mathbf{s}'$. \end{enumerate} An optional LLM-as-Judge scoring module provides multi-dimensional quality scores that can be used to filter low-quality samples. \subsection{DPO Data Generation} To construct pairwise preference data for Direct Preference Optimization, we start from a bilingual (Chinese--English) instruction set curated by the same MinHash pipeline described above. For each prompt $x$ in the instruction set, we generate two responses: \begin{itemize}[nosep] \item \textbf{Chosen} $y_w$: generated by the reference model $\pi_{\text{ref}}$, i.e.~the SFT checkpoint after supervised fine-tuning on the curated instruction data. \item \textbf{Rejected} $y_l$: generated by the base model $\pi_{\text{base}}$, i.e.~the model at the end of pretraining before any instruction tuning. \end{itemize} Both generations use greedy decoding to eliminate sampling variance and ensure that the preference signal reflects model capability rather than decoding randomness. The resulting preference pairs $(x, y_w, y_l)$ are stored in the same JSONL format used for SFT and fed directly into the DPO training loop (Section~\ref{sec:dpo}). Because the base model and the reference model share the same architecture but differ in instruction-following ability, the contrast between $y_w$ and $y_l$ is sharp and consistent, which stabilises the DPO gradient updates. All instructions are balanced across Chinese and English domains to preserve bilingual alignment capability during preference optimisation. \section{Model Architecture} % ====================================================================== The model is a 24-layer decoder-only Transformer with Grouped Query Attention (GQA)~\cite{ainslie2023gqa}, SwiGLU feed-forward blocks~\cite{shazeer2020glu}, and Rotary Position Embedding (RoPE)~\cite{su2024roformer}. Table~\ref{tab:model_config} summarizes the configuration. \begin{table}[H] \centering \caption{Model configuration. Total: $\sim$1.2B parameters.} \label{tab:model_config} \begin{tabular}{@{}lrlr@{}} \toprule \textbf{Parameter} & \textbf{Value} & \textbf{Parameter} & \textbf{Value} \\ \midrule Vocabulary ($V$) & 100,000 & Hidden dim ($d$) & 1,536 \\ Layers ($L$) & 24 & FFN dim ($d_{\textit{ffn}}$) & 6,912 \\ Query heads & 24 & KV heads & 4 \\ Head dim & 64 & Max length & 2,048 \\ Norm & RMSNorm ($\epsilon=10^{-5}$) & RoPE $\theta$ & 10,000 \\ \bottomrule \end{tabular} \end{table} With Grouped Query Attention~\cite{ainslie2023gqa} ($n_q = 24$ query heads, $n_{kv} = 4$ key/value heads, group size $g = n_q / n_{kv} = 6$): \begin{equation} \begin{aligned} \operatorname{GQA}(\mathbf{X}) &= \operatorname{Concat}\bigl(\operatorname{head}_1,\dots,\operatorname{head}_{n_q}\bigr)\mathbf{W}_O,\\[2mm] \operatorname{head}_i &= \operatorname{Attn}\Bigl( \mathbf{X}\mathbf{W}_Q^{(i)},\, \mathbf{X}\mathbf{W}_K^{(\lfloor i / g \rfloor)},\, \mathbf{X}\mathbf{W}_V^{(\lfloor i / g \rfloor)} \Bigr), \end{aligned} \end{equation} where $\operatorname{Attn}(\mathbf{Q},\mathbf{K},\mathbf{V}) = \operatorname{Softmax}(\mathbf{Q}\mathbf{K}^{\mkern-1mu\mathsf{T}} / \sqrt{d_h})\mathbf{V}$. Rotary Position Embedding (RoPE)~\cite{su2024roformer} encodes position $m$ by rotating pairs of hidden dimensions: \begin{equation} \operatorname{RoPE}(\mathbf{x}_m)_i = \begin{cases} x_{m,i}\cos(m\theta_{j}) - x_{m,i+1}\sin(m\theta_{j}), & i = 2j,\\[2mm] x_{m,i-1}\sin(m\theta_{j}) + x_{m,i}\cos(m\theta_{j}), & i = 2j+1, \end{cases} \end{equation} with frequency $\theta_j = 10000^{-2j/d}$ for $j = 0,\dots,d/2-1$. The SwiGLU~\cite{shazeer2020glu} feed-forward applies a gated Swish non-linearity: \begin{equation} \operatorname{MLP}(\mathbf{x}) = \mathbf{W}_{\text{down}}\Bigl( \mathbf{W}_{\text{up}}\mathbf{x} \odot \operatorname{SiLU}\bigl(\mathbf{W}_{\text{gate}}\mathbf{x}\bigr) \Bigr), \end{equation} where $\operatorname{SiLU}(z) = z / (1 + e^{-z})$. Each decoder block $\ell$ then applies pre-norm residual connections: \begin{equation} \begin{aligned} \mathbf{h}_\ell &= \mathbf{x}_\ell + \operatorname{GQA}\bigl(\operatorname{RMSNorm}(\mathbf{x}_\ell)\bigr),\\[2mm] \mathbf{x}_{\ell+1} &= \mathbf{h}_\ell + \operatorname{MLP}\bigl(\operatorname{RMSNorm}(\mathbf{h}_\ell)\bigr). \end{aligned} \end{equation} \subsection{Initialization} Linear weights follow $\mathcal{N}(0, 0.02)$; embeddings follow $\mathcal{N}(0, 0.02)$. The output projection $\mathbf{W}_o$ and FFN down-projection $\mathbf{W}_{\text{down}}$ use residual-scaled initialization~\cite{radford2019gpt2}: \begin{equation} \sigma_o = \sigma_{\text{down}} = 0.02 / \sqrt{2L}. \end{equation} This scaling is critical for BF16 stability (Section~\ref{sec:num-stability}). % ====================================================================== \section{Training Configuration} % ====================================================================== The model is trained on next-token cross-entropy loss: \begin{equation} \mathcal{L} = -\sum_{t=1}^{T} \log P(x_t \mid x_{ 1.0$. \item \textbf{1K SFT}: mean IFD $= 0.8485$, median $= 0.9083$, std $= 0.1588$; $3.1\%$ of samples exceed $1.0$. \item \textbf{Stability}: Pearson $r > 0.97$ between base and 1K SFT IFD. The slight upward shift ($0.8263 \to 0.8485$) reflects both losses increasing after SFT, consistent with distribution shift during fine-tuning rather than uniform instruction-following improvement. \end{itemize} \subsection{Representative Samples} Table~\ref{tab:ifd_examples} lists samples spanning the IFD range. \begin{table}[H] \centering \caption{Representative IFD samples.} \label{tab:ifd_examples} \small \begin{tabular}{@{}c c c c c p{4.2cm}@{}} \toprule \textbf{Idx} & \textbf{$L_{\text{cond}}^{\text{base}}$} & \textbf{$L_{\text{uncond}}^{\text{base}}$} & \textbf{$L_{\text{cond}}^{\text{1K}}$} & \textbf{$L_{\text{uncond}}^{\text{1K}}$} & \textbf{Instruction} \\ \midrule 81 & 13.38 & 5.84 & 13.25 & 5.69 & Classify incident as breach of protocol \\ 906 & 13.12 & 9.75 & 13.06 & 9.75 & Convert numbers from words to digits \\ 1076 & 2.53 & 2.46 & 2.53 & 2.53 & Pick best synonym \\ 7 & 2.62 & 2.70 & 2.68 & 2.77 & Write a short story in third person \\ 2427 & 2.59 & 2.84 & 2.69 & 2.90 & Find five most similar sentences \\ 798 & 2.02 & 2.75 & 2.11 & 2.31 & List four social media platforms \\ 223 & 1.34 & 3.16 & 1.36 & 3.27 & Classify text as Fiction or Non-fiction \\ \bottomrule \end{tabular} \end{table} Samples with the highest conditional loss (rows~81,~906) are short-answer classification tasks ($L_{\text{cond}} \approx 13$). Lowest-IFD samples (row~223) are tasks where the instruction constrains the output space so tightly that unconditional loss far exceeds conditional loss. The four loss values remain nearly unchanged after SFT across all samples. \subsection{IFD Bias from Response Length} \label{sec:ifd_bias} Both losses are per-token averages. The variance of $L_{\text{uncond}} = \frac{1}{T} \sum_{t=1}^T \log P(x_t)$ scales as $1/T$, so shorter responses produce noisier estimates. Figure~\ref{fig:length_bias} plots the three metrics against response length for the base model; samples with $<20$ tokens ($21.9\%$ of the dataset) exhibit substantially higher scatter. \begin{figure}[H] \centering \includegraphics[width=0.95\linewidth]{data/ifd_length_grid.png} \caption{Response length vs.\ $L_{\text{cond}}$, $L_{\text{uncond}}$, and IFD (base model, log scale on $x$-axis).} \label{fig:length_bias} \end{figure} Table~\ref{tab:corr_bias} reports the correlations. Response length is the dominant confound: $L_{\text{uncond}}$ shows a strong negative monotonic trend ($\rho = -0.79$), while $L_{\text{cond}}$ is less affected ($\rho = -0.48$). The net effect on IFD is a positive correlation ($\rho = +0.72$). \begin{table}[H] \centering \caption{Pearson $r$ and Spearman $\rho$ between sample dimensions and IFD components (base model).} \label{tab:corr_bias} \small \begin{tabular}{@{}lcccccc@{}} \toprule & \multicolumn{2}{c}{vs.\ $L_{\text{cond}}$} & \multicolumn{2}{c}{vs.\ $L_{\text{uncond}}$} & \multicolumn{2}{c}{vs.\ IFD} \\ \cmidrule(lr){2-3} \cmidrule(lr){4-5} \cmidrule(lr){6-7} \textbf{Dimension} & $r$ & $\rho$ & $r$ & $\rho$ & $r$ & $\rho$ \\ \midrule Instruction length & $+0.07$ & $+0.06$ & $+0.15$ & $+0.24$ & $-0.25$ & $-0.34$ \\ Response length & $-0.36$ & $-0.48$ & $-0.56$ & $-0.79$ & $+0.58$ & $+0.72$ \\ \bottomrule \end{tabular} \end{table} \subsection{Loss Ratio} \label{sec:loss_ratio} We further define the \textbf{Loss Ratio} as the fraction of conditional loss retained after SFT: \begin{equation} \text{Loss Ratio} = \frac{L_{\text{cond}}^{\text{1K}}}{L_{\text{cond}}^{\text{base}}}. \end{equation} Over $N=3000$ samples: \begin{itemize}[nosep] \item Mean $= 1.106$, median $= 1.084$, std $= 0.110$; $90.8\%$ of samples exceed $1.0$. \item Range: $[0.768, 1.962]$; only $9.2\%$ of samples show a decrease ($<1.0$) in conditional loss after SFT. \end{itemize} The predominance of loss ratio $>1$ confirms that the 1K-step SFT checkpoint has not converged to a lower-loss region for the evaluation samples. Instead, the fine-tuning distribution shift increases NLL on most held-out instructions. Table~\ref{tab:ifd_lr_corr} reports the pairwise correlations. Although IFD\textsubscript{ckpt} and Loss Ratio both depend on $L_{\text{cond}}^{\text{1K}}$, they need not correlate because their denominators vary independently across samples. The observed correlation is near zero ($r = -0.02$, $\rho = 0.04$), precisely because the {\em relative} ordering of $L_{\text{cond}}^{\text{base}}$ and $L_{\text{uncond}}^{\text{ckpt}}$ (which determine the slope $k_i = L_{\text{cond},i}^{\text{base}} / L_{\text{uncond},i}^{\text{ckpt}}$ in the relationship $\text{IFD}_{\text{ckpt},i} = k_i \cdot \text{Loss Ratio}_i$) varies widely, breaking the proportionality at the sample level. This invalidates the naive expectation that a shared numerator guarantees correlation~\cite{li2023ifd}. \begin{table}[H] \centering \caption{Pairwise correlations between IFD variants and Loss Ratio.} \label{tab:ifd_lr_corr} \small \begin{tabular}{@{}lcc@{}} \toprule \textbf{Pair} & Pearson $r$ & Spearman $\rho$ \\ \midrule IFD\textsubscript{base} vs.\ IFD\textsubscript{ckpt} & $+0.97$ & $+0.96$ \\ IFD\textsubscript{base} vs.\ Loss Ratio & $-0.15$ & $-0.06$ \\ IFD\textsubscript{ckpt} vs.\ Loss Ratio & $-0.02$ & $+0.04$ \\ \bottomrule \end{tabular} \end{table} The near-perfect correlation between IFD\textsubscript{base} and IFD\textsubscript{ckpt} ($r = 0.97$) reveals that the IFD ranking is highly robust to the choice of evaluation model: samples that the base model finds difficult remain difficult after 1K SFT steps. This stability justifies using the base-model IFD as a data selection signal without re-evaluating after fine-tuning. % ====================================================================== \section{Effective Rank Analysis} \label{app:eff_rank} % ====================================================================== To assess how well the trained parameters utilize their allocated capacity, we perform an SVD-based effective rank analysis on three checkpoints: \texttt{kami-15bt} (AdamW, GPT-2 residual scaling, 15B tokens), \texttt{norm-15bt} (Normal init, 15B tokens), and \texttt{muon-25bt} (Muon, 25B tokens). For each 2D weight matrix $\mathbf{W} \in \mathbb{R}^{m\times n}$ with SVD $\mathbf{W} = \mathbf{U}\boldsymbol{\Sigma}\mathbf{V}^{\mkern-1mu\mathsf{T}}$, we compute the effective rank at 99\% energy: \begin{equation} \text{ER@99\%} = \frac{1}{\min(m,n)} \min_k \left\{ k \;\middle|\; \frac{\sum_{i=1}^{k} \sigma_i^2}{\sum_{i=1}^{\min(m,n)} \sigma_i^2} \ge 0.99 \right\}. \end{equation} Table~\ref{tab:eff_rank} summarizes the results. All three checkpoints exhibit a high overall ER@99\% ($\sim$90\%), indicating that the 1.2B model operates close to its representational capacity. Key findings: \begin{itemize}[nosep] \item Q/O projections show lower ER@99\% ($\sim$0.73--0.77) and high condition numbers ($\kappa > 10^4$), consistent with the low-rank structure of GQA (24 query heads sharing 4 KV heads). \item K/V projections, FFN layers, and embeddings maintain high ER@99\% ($\sim$0.96--0.98) and low condition numbers ($\kappa < 10$). \item The overall ER@99\% varies by less than 0.01 across checkpoints, indicating these properties are determined primarily by architecture rather than optimizer or training duration. \end{itemize} \begin{table}[H] \centering \caption{SVD effective rank (ER@99\%) and mean condition number ($\kappa$) by component across three checkpoints.} \label{tab:eff_rank} \small \begin{tabular}{@{}lcccccc@{}} \toprule & \multicolumn{3}{c}{\textbf{ER@99\%}} & \multicolumn{3}{c}{\textbf{Cond.\ Number $\kappa$}} \\ \cmidrule(lr){2-4} \cmidrule(lr){5-7} \textbf{Component} & \textbf{kami} & \textbf{norm} & \textbf{muon} & \textbf{kami} & \textbf{norm} & \textbf{muon} \\ \midrule attn.k\_proj & 0.967 & 0.971 & 0.960 & 8.4 & 5.4 & 5.7 \\ attn.o\_proj & 0.766 & 0.708 & 0.730 & 31{,}589 & 57{,}284 & 22{,}838 \\ attn.q\_proj & 0.754 & 0.764 & 0.756 & 46{,}644 & 32{,}072 & 14{,}472 \\ attn.v\_proj & 0.976 & 0.976 & 0.971 & 2.5 & 2.4 & 2.9 \\ embed\_tokens & 0.985 & 0.987 & 0.984 & 4.9 & 1.9 & 3.5 \\ lm\_head & 0.969 & 0.981 & 0.980 & 21.0 & 13.5 & 16.1 \\ mlp.down & 0.961 & 0.961 & 0.963 & 6.4 & 7.2 & 7.2 \\ mlp.gate & 0.966 & 0.967 & 0.965 & 5.7 & 5.3 & 5.7 \\ mlp.up & 0.967 & 0.968 & 0.965 & 4.9 & 5.9 & 4.9 \\ \midrule \textbf{Overall ER@99\%} & \textbf{0.909} & \textbf{0.903} & \textbf{0.903} & & & \\ \bottomrule \end{tabular} \end{table} % ====================================================================== \section{Per-Component Weight Statistics} \label{app:weight_std} % ====================================================================== Table~\ref{tab:weight_std} reports the weight standard deviation by component for each checkpoint, supplementing the post-training weight distribution analysis in Section~\ref{sec:num-stability}. \begin{table}[H] \centering \caption{Weight std by component across checkpoints. Non-scaled weights broaden with training; residual scaling constrains drift at equal token count (15B). Muon produces larger post-convergence weight variance than AdamW. Residual-scaled projections ($\mathbf{W}_o$, $\mathbf{W}_{\text{down}}$) remain bounded.} \label{tab:weight_std} \small \begin{tabular}{@{}lccc@{}} \toprule \textbf{Component} & \textbf{kami-15bt} & \textbf{norm-15bt} & \textbf{muon-25bt} \\ & (AdamW, 15B) & (AdamW, 15B) & (Muon, 25B) \\ \midrule attn.q\_proj & 0.0154 & 0.0207 & 0.0230 \\ attn.k\_proj & 0.0153 & 0.0206 & 0.0238 \\ attn.v\_proj & 0.0146 & 0.0202 & 0.0244 \\ attn.o\_proj$^*$ & 0.0148 & 0.0084 & 0.0177 \\ mlp.up & 0.0153 & 0.0204 & 0.0237 \\ mlp.gate & 0.0155 & 0.0204 & 0.0235 \\ mlp.down$^*$ & 0.0100 & 0.0089 & 0.0180 \\ embed\_tokens & 0.0205 & 0.0205 & 0.0239 \\ lm\_head & 0.0224 & 0.0257 & 0.0298 \\ \bottomrule \end{tabular} \\[2pt] \footnotesize $^*$Residual-scaled projection ($\sigma_0 = 0.02/\sqrt{2L}$). \end{table} % ====================================================================== \begin{thebibliography}{99} \bibitem{alpaca} R.~Taori, I.~Gulrajani, T.~Zhang, Y.~Dubois, X.~Li, C.~Guestrin, P.~Liang, T.~B.~Hashimoto. Alpaca: A strong, replicable instruction-following model. \textit{Stanford Center for Research on Foundation Models (CRFM)}, 2023. \bibitem{ainslie2023gqa} J.~Ainslie, J.~Lee-Thorp, M.~de Jong, Y.~Zemlyanskiy, F.~Lebr\'on, S.~Sanghai. GQA: Training generalized multi-query transformer models from multi-head checkpoints. \textit{EMNLP}, 2023. \bibitem{alembic} Alembic Contributors. \textit{Alembic: A lightweight LLM-driven SFT data generation, cleaning, and scoring pipeline.} \url{https://github.com/ViperEkura/Alembic}, 2026. \bibitem{astrai} AstrAI Contributors. \textit{AstrAI: An open-source training and inference framework for Transformer language models.} \url{https://github.com/ViperEkura/AstrAI}, 2026. \bibitem{broder1997syntactic} A.~Z.~Broder. On the resemblance and containment of documents. \textit{SEQUENCES '97}, 1997. \bibitem{li2023ifd} M.~Li, Y.~Zhang, Z.~Li, J.~Chen, L.~Chen, N.~Cheng, J.~Wang, T.~Zhou, J.~Xiao. From quantity to quality: Boosting LLM performance with self-guided data selection for instruction tuning. \textit{NAACL}, 2024. \bibitem{ieee754} IEEE Computer Society. \textit{IEEE Standard for Floating-Point Arithmetic}, IEEE Std 754-2019, 2019. \bibitem{loshchilov2019adamw} I.~Loshchilov, F.~Hutter. Decoupled weight decay regularization. \textit{ICLR}, 2019. \bibitem{radford2019gpt2} A.~Radford, J.~Wu, R.~Child, D.~Luan, D.~Amodei, I.~Sutskever. Language models are unsupervised multitask learners. \textit{OpenAI Blog}, 2019. \bibitem{rafailov2023dpo} R.~Rafailov, A.~Sharma, E.~Mitchell, C.~D.~Manning, S.~Ermon, C.~Finn. Direct Preference Optimization: Your language model is secretly a reward model. \textit{NeurIPS}, 2023. \bibitem{shazeer2020glu} N.~Shazeer. GLU variants improve Transformer. \textit{arXiv:2002.05202}, 2020. \bibitem{su2024roformer} J.~Su, A.~Murtadha, Y.~Lu, S.~Pan, B.~Wen, Y.~Liu. Roformer: Enhanced transformer with rotary position embedding. \textit{Neurocomputing}, 568:127063, 2024. \end{thebibliography} \end{document}