{"id":3539,"date":"2025-11-25T11:12:57","date_gmt":"2025-11-25T10:12:57","guid":{"rendered":"https:\/\/neuraldesigner.com\/learning\/training-strategy\/"},"modified":"2026-08-26T10:53:57","modified_gmt":"2026-08-26T08:53:57","slug":"training-strategy","status":"publish","type":"learning","link":"https:\/\/www.neuraldesigner.com\/learning\/tutorials\/training-strategy\/","title":{"rendered":"Machine learning: Training strategy &#8211; tutorial"},"content":{"rendered":"<style>\n.nd-math-block{display:block;max-width:100%;overflow-x:auto;margin:1rem 0;padding:.45rem 0;text-align:center}.nd-math-inline{display:inline-block;max-width:100%;overflow-x:auto;vertical-align:middle}.nd-math-block math,.nd-math-inline math{font-size:1.04em}table .nd-math-block{margin:.25rem 0;padding:.2rem 0}\n<\/style>\n<style>.ndg{width:100vw;margin-left:calc(50% - 50vw);background:#eeeeee;padding:22px 24px 12px;font-family:\"Outfit\",\"Roboto\",Arial,sans-serif;color:#1b2635}.ndg *{box-sizing:border-box}.ndg a{text-decoration:none}.ndg-wrap{width:min(100%,1000px);margin:0 auto}.ndg-lead{font-size:19px;line-height:1.6;color:#3a4a5a;font-weight:300;margin:0 0 24px}.ndg-lead a{color:#2d799f;font-weight:600}.ndg-herofig{text-align:center;margin:8px 0 22px}.ndg-herofig img{display:inline-block;width:64px;height:64px}.ndg-eyebrow{margin:0 0 14px;text-align:center;color:#2d799f;font-size:13px;font-weight:800;letter-spacing:.14em;text-transform:uppercase}.ndg-toc{list-style:none;counter-reset:s;display:grid;grid-template-columns:repeat(3,minmax(0,1fr));gap:12px;margin:0 0 48px;padding:0}.ndg-toc li{counter-increment:s}.ndg-toc a{display:flex;align-items:center;gap:12px;height:100%;padding:13px 16px;background:#f2f2f2;border-radius:12px;color:#12354b!important;font-size:14.5px;font-weight:600;line-height:1.25;box-shadow:-8px -8px 16px rgba(255,255,255,.9),8px 8px 16px rgba(30,83,116,.10)}.ndg-toc a:hover{color:#2d799f!important}.ndg-toc a:before{content:counter(s);display:flex;align-items:center;justify-content:center;width:26px;height:26px;flex:0 0 26px;border-radius:50%;background:#56a1c8;color:#fff;font-size:13px;font-weight:800}.ndg-step{display:grid;grid-template-columns:64px 1fr;gap:26px;margin:0 0 34px;padding:34px 38px;background:#f2f2f2;border-radius:22px;box-shadow:-14px -14px 28px rgba(255,255,255,.92),14px 14px 28px rgba(30,83,116,.12);scroll-margin-top:90px}.ndg-step__no{display:flex;align-items:center;justify-content:center;width:64px;height:64px;border-radius:50%;background:linear-gradient(135deg,#56a1c8 0%,#245e80 100%);color:#fff;font-size:25px;font-weight:800;flex:0 0 64px}.ndg-step__body{min-width:0}.ndg-step__body .mjx-chtml.MJXc-display{overflow-x:auto;overflow-y:hidden;max-width:100%;padding:2px 0 8px}.ndg-step__body .mjx-chtml.MathJax_CHTML{font-size:18px!important}.ndg-step__body h2{margin:8px 0 16px;color:#001233;font-size:25px;font-weight:700;line-height:1.2}.ndg-step__body h3{margin:28px 0 12px;color:#12354b;font-size:19px;font-weight:600;scroll-margin-top:90px}.ndg-anchor-alias{display:block;height:0;position:relative;top:-90px;visibility:hidden}.ndg-step__body p{margin:0 0 16px;font-size:17px;line-height:1.62;color:#33424f}.ndg-step__body a{color:#2d799f;font-weight:500}.ndg-step__body ul{margin:0 0 16px;padding-left:22px}.ndg-step__body li{margin:7px 0;font-size:16px;line-height:1.55;color:#33424f}.ndg-step__body img:not([src$=\".svg\"]):not([data-src$=\".svg\"]){display:block!important;width:auto!important;max-width:min(580px,100%)!important;height:auto!important;margin:22px auto!important;border-radius:12px!important;box-shadow:0 12px 28px rgba(0,18,51,.12)!important}.ndg-step__body img[src$=\".svg\"],.ndg-step__body img[data-src$=\".svg\"]{display:block;margin:18px auto;max-width:min(520px,100%);height:auto}.ndg-note{margin:18px 0;padding:15px 18px;border-left:4px solid #56a1c8;background:rgba(86,161,200,.08);border-radius:0 10px 10px 0;color:#33424f;font-size:16px;line-height:1.55}.ndg-table-wrap{max-width:100%;overflow-x:auto;margin:18px 0 22px}.ndg-table{width:100%;min-width:620px;border-collapse:separate;border-spacing:0;background:#fff;border:1px solid #c8d7df;border-radius:12px;overflow:hidden}.ndg-table th,.ndg-table td{padding:12px 14px;border-bottom:1px solid #dce5ea;text-align:left;vertical-align:top;font-size:15px;line-height:1.45;color:#33424f}.ndg-table th{background:#e8f1f6;color:#12354b;font-weight:700}.ndg-table tr:last-child td{border-bottom:0}.ndg-nav{display:flex;justify-content:space-between;gap:16px;margin:12px 0 8px;flex-wrap:wrap}.ndg-nav a{display:inline-flex;align-items:center;padding:13px 24px;border-radius:10px;background:#f2f2f2;color:#2d799f!important;font-weight:800;font-size:15px;box-shadow:-8px -8px 16px rgba(255,255,255,.92),8px 8px 16px rgba(30,83,116,.10)}.ndg-nav a:hover{color:#1f5f80!important}@media(max-width:820px){.ndg-toc{grid-template-columns:1fr}.ndg-step{grid-template-columns:1fr;gap:16px;padding:26px 22px}.ndg-step__no{width:52px;height:52px;flex:0 0 52px;font-size:22px}.ndg-step__body h2{font-size:22px}}@media(max-width:640px){.ndg{padding:12px 14px}.ndg-table th,.ndg-table td{padding:10px 12px}}<\/style>\n<div class=\"ndg\"><div class=\"ndg-wrap\"><div class=\"ndg-lead\">\n<p>Tutorial index:<\/p>\n<ul>\n<li><a href=\"https:\/\/www.neuraldesigner.com\/learning\/tutorials\/neural-networks-applications\/\">1. Model types<\/a><\/li>\n<li><a href=\"https:\/\/www.neuraldesigner.com\/learning\/tutorials\/data-set\/\">2. Data set<\/a><\/li>\n<li><a href=\"https:\/\/www.neuraldesigner.com\/learning\/tutorials\/neural-network\/\">3. Neural network<\/a><\/li>\n<li><a href=\"https:\/\/www.neuraldesigner.com\/learning\/tutorials\/training-strategy\/\">4. Training strategy<\/a><\/li>\n<li><a href=\"https:\/\/www.neuraldesigner.com\/learning\/tutorials\/model-selection\/\">5. Model selection<\/a><\/li>\n<li><a href=\"https:\/\/www.neuraldesigner.com\/learning\/tutorials\/testing-analysis\/\">6. Testing analysis<\/a><\/li>\n<li><a href=\"https:\/\/www.neuraldesigner.com\/learning\/tutorials\/model-deployment\/\">7. Model deployment<\/a><\/li>\n<\/ul>\n<h2>4. Training strategy<\/h2>\n<p>The training strategy determines how a neural network learns from a data set. It combines a loss, which measures the model error, with an optimization algorithm, which updates the model parameters.<\/p>\n<p>OpenNN selects suitable defaults from the neural network task. The loss, optimizer, stopping criteria, batch processing, and validation behavior can also be configured.<\/p>\n<\/div><div class=\"ndg-herofig\"><img decoding=\"async\" src=\"https:\/\/www.neuraldesigner.com\/images\/training_strategy.svg\" width=\"64\" height=\"64\" alt=\"Training strategy\" \/><\/div>\n<p class=\"ndg-eyebrow\">Contents<\/p>\n<ul class=\"ndg-toc\"><li><a href=\"#Loss\">Loss<\/a><\/li><li><a href=\"#OptimizationAlgorithms\">Optimization algorithms<\/a><\/li><li><a href=\"#TrainingControl\">Training control and defaults<\/a><\/li><\/ul>\n<div class=\"ndg-step\" id=\"Loss\"><div class=\"ndg-step__no\">1<\/div><div class=\"ndg-step__body\"><span id=\"LossIndex\" class=\"ndg-anchor-alias\" aria-hidden=\"true\"><\/span><h2>Loss<\/h2>\n<p>The loss defines what the neural network must learn. It combines an error term with an optional regularization term:<\/p>\n<p><span class=\"nd-math-block\"><math xmlns=\"http:\/\/www.w3.org\/1998\/Math\/MathML\" display=\"block\" aria-label=\"loss equals error plus regularization\"><mrow><mtext>loss<\/mtext><mo>=<\/mo><mtext>error<\/mtext><mo>+<\/mo><mtext>regularization<\/mtext><\/mrow><\/math><\/span><\/p>\n<h3 id=\"ErrorTerm\">Error term<\/h3>\n<p>The error measures the difference between the outputs from the <a href=\"https:\/\/www.neuraldesigner.com\/learning\/tutorials\/neural-network\/\">neural network<\/a> and the targets in the <a href=\"https:\/\/www.neuraldesigner.com\/learning\/tutorials\/data-set\/\">data set<\/a>.<\/p>\n<p>The training error is calculated on the <a href=\"https:\/\/www.neuraldesigner.com\/learning\/tutorials\/data-set\/#TrainingSamples\">training samples<\/a>. The validation error monitors generalization on the <a href=\"https:\/\/www.neuraldesigner.com\/learning\/tutorials\/data-set\/#SelectionSamples\">validation samples<\/a>. The <a href=\"https:\/\/www.neuraldesigner.com\/learning\/tutorials\/data-set\/#TestingSamples\">testing samples<\/a> are reserved for the final evaluation and do not update the model.<\/p>\n<p>OpenNN provides the following general-purpose error methods:<\/p>\n<ul>\n<li><a href=\"#MeanSquaredError\">Mean squared error<\/a>.<\/li>\n<li><a href=\"#MeanAbsoluteError\">Mean absolute error<\/a>.<\/li>\n<li><a href=\"#NormalizedSquaredError\">Normalized squared error<\/a>.<\/li>\n<li><a href=\"#WeightedSquaredError\">Weighted squared error<\/a>.<\/li>\n<li><a href=\"#CrossEntropyError\">Cross-entropy error<\/a>.<\/li>\n<li><a href=\"#CrossEntropyError3d\">3D cross-entropy error<\/a>.<\/li>\n<li><a href=\"#MinkowskiError\">Minkowski error<\/a>.<\/li>\n<\/ul>\n<h3 id=\"MeanSquaredError\">Mean squared error (MSE)<\/h3>\n<p>The mean squared error penalizes large differences between outputs and targets. It is the default error for approximation, forecasting, and auto-association.<\/p>\n<p><span class=\"nd-math-block\"><math xmlns=\"http:\/\/www.w3.org\/1998\/Math\/MathML\" display=\"block\" aria-label=\"mean squared error\"><mrow><mtext>MSE<\/mtext><mo>=<\/mo><mfrac><mn>1<\/mn><mrow><mn>2<\/mn><mi>N<\/mi><\/mrow><\/mfrac><munderover><mo>\u2211<\/mo><mrow><mi>i<\/mi><mo>=<\/mo><mn>1<\/mn><\/mrow><mi>N<\/mi><\/munderover><msup><mrow><mo>\u2016<\/mo><msub><mi>y<\/mi><mi>i<\/mi><\/msub><mo>\u2212<\/mo><msub><mi>t<\/mi><mi>i<\/mi><\/msub><mo>\u2016<\/mo><\/mrow><mn>2<\/mn><\/msup><\/mrow><\/math><\/span><\/p>\n<h3 id=\"MeanAbsoluteError\">Mean absolute error (MAE)<\/h3>\n<p>The mean absolute error averages the absolute differences between outputs and targets. It is less sensitive to large individual errors than squared losses.<\/p>\n<p><span class=\"nd-math-block\"><math xmlns=\"http:\/\/www.w3.org\/1998\/Math\/MathML\" display=\"block\" aria-label=\"mean absolute error\"><mrow><mtext>MAE<\/mtext><mo>=<\/mo><mfrac><mn>1<\/mn><mi>M<\/mi><\/mfrac><munderover><mo>\u2211<\/mo><mrow><mi>j<\/mi><mo>=<\/mo><mn>1<\/mn><\/mrow><mi>M<\/mi><\/munderover><mrow><mo>|<\/mo><msub><mi>y<\/mi><mi>j<\/mi><\/msub><mo>\u2212<\/mo><msub><mi>t<\/mi><mi>j<\/mi><\/msub><mo>|<\/mo><\/mrow><\/mrow><\/math><\/span><\/p>\n<h3 id=\"NormalizedSquaredError\">Normalized squared error (NSE)<\/h3>\n<p>The normalized squared error divides the squared error by the target variance. A value close to one represents prediction around the target mean, while zero represents a perfect prediction.<\/p>\n<p><span class=\"nd-math-block\"><math xmlns=\"http:\/\/www.w3.org\/1998\/Math\/MathML\" display=\"block\" aria-label=\"normalized squared error\"><mrow><mtext>NSE<\/mtext><mo>=<\/mo><mfrac><mrow><mo>\u2211<\/mo><msup><mrow><mo>\u2016<\/mo><mi>y<\/mi><mo>\u2212<\/mo><mi>t<\/mi><mo>\u2016<\/mo><\/mrow><mn>2<\/mn><\/msup><\/mrow><mrow><mo>\u2211<\/mo><msup><mrow><mo>\u2016<\/mo><mi>t<\/mi><mo>\u2212<\/mo><mover><mi>t<\/mi><mo>\u00af<\/mo><\/mover><mo>\u2016<\/mo><\/mrow><mn>2<\/mn><\/msup><\/mrow><\/mfrac><\/mrow><\/math><\/span><\/p>\n<h3 id=\"WeightedSquaredError\">Weighted squared error (WSE)<\/h3>\n<p>The weighted squared error is designed for imbalanced binary classification. OpenNN calculates class weights from the positive and negative training samples and gives both classes a balanced contribution.<\/p>\n<p><span class=\"nd-math-block\"><math xmlns=\"http:\/\/www.w3.org\/1998\/Math\/MathML\" display=\"block\" aria-label=\"weighted squared error\"><mrow><mtext>WSE<\/mtext><mo>\u221d<\/mo><mfrac><mn>1<\/mn><mn>2<\/mn><\/mfrac><munderover><mo>\u2211<\/mo><mrow><mi>i<\/mi><mo>=<\/mo><mn>1<\/mn><\/mrow><mi>N<\/mi><\/munderover><msub><mi>w<\/mi><msub><mi>t<\/mi><mi>i<\/mi><\/msub><\/msub><msup><mrow><mo>(<\/mo><msub><mi>y<\/mi><mi>i<\/mi><\/msub><mo>\u2212<\/mo><msub><mi>t<\/mi><mi>i<\/mi><\/msub><mo>)<\/mo><\/mrow><mn>2<\/mn><\/msup><\/mrow><\/math><\/span><\/p>\n<h3 id=\"CrossEntropyError\">Cross-entropy error<\/h3>\n<p>Cross-entropy measures probabilistic classification. OpenNN uses binary cross-entropy for one output and categorical cross-entropy for multiple outputs. Multiclass models require a softmax output.<\/p>\n<p><span class=\"nd-math-block\"><math xmlns=\"http:\/\/www.w3.org\/1998\/Math\/MathML\" display=\"block\" aria-label=\"binary cross entropy\"><mrow><mtext>BCE<\/mtext><mo>=<\/mo><mo>\u2212<\/mo><mfrac><mn>1<\/mn><mi>N<\/mi><\/mfrac><mo>\u2211<\/mo><mrow><mo>[<\/mo><mi>t<\/mi><mi>log<\/mi><mo>(<\/mo><mi>y<\/mi><mo>)<\/mo><mo>+<\/mo><mo>(<\/mo><mn>1<\/mn><mo>\u2212<\/mo><mi>t<\/mi><mo>)<\/mo><mi>log<\/mi><mo>(<\/mo><mn>1<\/mn><mo>\u2212<\/mo><mi>y<\/mi><mo>)<\/mo><mo>]<\/mo><\/mrow><\/mrow><\/math><\/span><\/p>\n<h3 id=\"CrossEntropyError3d\">3D cross-entropy error<\/h3>\n<p>The 3D cross-entropy error trains language models and other token sequences. It averages categorical cross-entropy over active tokens and also reports token accuracy and perplexity.<\/p>\n<h3 id=\"MinkowskiError\">Minkowski error (ME)<\/h3>\n<p>The Minkowski error uses a power between one and two to reduce the influence of large residuals. Its default power is 1.5. This loss currently runs on CPU.<\/p>\n<p><span class=\"nd-math-block\"><math xmlns=\"http:\/\/www.w3.org\/1998\/Math\/MathML\" display=\"block\" aria-label=\"Minkowski error\"><mrow><mtext>ME<\/mtext><mo>=<\/mo><mfrac><mn>1<\/mn><mrow><mi>p<\/mi><mi>N<\/mi><\/mrow><\/mfrac><munderover><mo>\u2211<\/mo><mrow><mi>i<\/mi><mo>=<\/mo><mn>1<\/mn><\/mrow><mi>N<\/mi><\/munderover><msup><mrow><mo>|<\/mo><msub><mi>y<\/mi><mi>i<\/mi><\/msub><mo>\u2212<\/mo><msub><mi>t<\/mi><mi>i<\/mi><\/msub><mo>|<\/mo><\/mrow><mi>p<\/mi><\/msup><\/mrow><\/math><\/span><\/p>\n<h3 id=\"RegularizationTerm\">Regularization term<\/h3>\n<p>Regularization penalizes large parameters to control model complexity. It is optional, and no regularization is applied by default.<\/p>\n<ul><li><a href=\"#L1Regularization\">L1 regularization<\/a>.<\/li><li><a href=\"#L2Regularization\">L2 regularization<\/a>.<\/li><\/ul>\n<h3 id=\"L1Regularization\">L1 regularization<\/h3>\n<p>L1 regularization adds the sum of the absolute parameter values to the loss.<\/p>\n<p><span class=\"nd-math-block\"><math xmlns=\"http:\/\/www.w3.org\/1998\/Math\/MathML\" display=\"block\" aria-label=\"L1 regularization\"><mrow><msub><mi>R<\/mi><mn>1<\/mn><\/msub><mo>=<\/mo><mi>\u03bb<\/mi><mo>\u2211<\/mo><mo>|<\/mo><mi>\u03b8<\/mi><mo>|<\/mo><\/mrow><\/math><\/span><\/p>\n<h3 id=\"L2Regularization\">L2 regularization<\/h3>\n<p>L2 regularization adds the sum of the squared parameter values to the loss.<\/p>\n<p><span class=\"nd-math-block\"><math xmlns=\"http:\/\/www.w3.org\/1998\/Math\/MathML\" display=\"block\" aria-label=\"L2 regularization\"><mrow><msub><mi>R<\/mi><mn>2<\/mn><\/msub><mo>=<\/mo><mi>\u03bb<\/mi><mo>\u2211<\/mo><msup><mi>\u03b8<\/mi><mn>2<\/mn><\/msup><\/mrow><\/math><\/span><\/p>\n<h3 id=\"LossFunction\">Loss function<\/h3>\n<p>The loss forms a surface over the neural network parameters. Training searches this surface for parameter values that minimize the selected objective.<\/p>\n<p><img decoding=\"async\" src=\"https:\/\/www.neuraldesigner.com\/images\/loss_function.svg\" alt=\"Loss function\" width=\"500\" \/><\/p>\n<\/div><\/div>\n<div class=\"ndg-step\" id=\"OptimizationAlgorithms\"><div class=\"ndg-step__no\">2<\/div><div class=\"ndg-step__body\"><span id=\"OptimizationAlgorithm\" class=\"ndg-anchor-alias\" aria-hidden=\"true\"><\/span><span id=\"ConjugateGradient\" class=\"ndg-anchor-alias\" aria-hidden=\"true\"><\/span><h2>Optimization algorithms<\/h2>\n<p>The optimization algorithm updates the neural network parameters to reduce the loss. OpenNN currently provides four algorithms:<\/p>\n<ul><li><a href=\"#AdaptiveMomentEstimation\">Adaptive moment estimation (Adam)<\/a>.<\/li><li><a href=\"#StochasticGradientDescent\">Stochastic gradient descent (SGD)<\/a>.<\/li><li><a href=\"#QuasiNewtonMethod\">Quasi-Newton method<\/a>.<\/li><li><a href=\"#LevenbergMarquardtAlgorithm\">Levenberg-Marquardt algorithm<\/a>.<\/li><\/ul>\n<div class=\"ndg-table-wrap\"><table class=\"ndg-table\"><thead><tr><th>Algorithm<\/th><th>Training mode<\/th><th>Hardware<\/th><th>Best suited for<\/th><\/tr><\/thead><tbody>\n<tr><td>Adam<\/td><td>Mini-batch<\/td><td>CPU and GPU<\/td><td>Large, image, text, and sequence models<\/td><\/tr>\n<tr><td>SGD<\/td><td>Mini-batch<\/td><td>CPU and GPU<\/td><td>Controlled learning schedules and large data sets<\/td><\/tr>\n<tr><td>Quasi-Newton<\/td><td>Full-batch<\/td><td>CPU<\/td><td>Tabular and moderate-size problems<\/td><\/tr>\n<tr><td>Levenberg-Marquardt<\/td><td>Full-batch<\/td><td>CPU<\/td><td>Small sequential Dense regression models<\/td><\/tr>\n<\/tbody><\/table><\/div>\n<h3 id=\"AdaptiveMomentEstimation\">Adaptive moment estimation (Adam)<\/h3><span id=\"AdaptativeLinearMomentum\" class=\"ndg-anchor-alias\" aria-hidden=\"true\"><\/span>\n<p>Adam combines exponential averages of the gradient and squared gradient with bias correction. Its main parameters are the learning rate, beta 1, and beta 2.<\/p>\n<p><span class=\"nd-math-block\"><math xmlns=\"http:\/\/www.w3.org\/1998\/Math\/MathML\" display=\"block\" aria-label=\"Adam parameter update\"><mrow><msub><mi>\u03b8<\/mi><mi>t<\/mi><\/msub><mo>=<\/mo><msub><mi>\u03b8<\/mi><mrow><mi>t<\/mi><mo>\u2212<\/mo><mn>1<\/mn><\/mrow><\/msub><mo>\u2212<\/mo><mi>\u03b1<\/mi><mfrac><msub><mover><mi>m<\/mi><mo>^<\/mo><\/mover><mi>t<\/mi><\/msub><mrow><msqrt><msub><mover><mi>v<\/mi><mo>^<\/mo><\/mover><mi>t<\/mi><\/msub><\/msqrt><mo>+<\/mo><mi>\u03b5<\/mi><\/mrow><\/mfrac><\/mrow><\/math><\/span><\/p>\n<p>Adam is the default optimizer for approximation, forecasting, auto-association, image classification, text classification, and language modeling.<\/p>\n<h3 id=\"StochasticGradientDescent\">Stochastic gradient descent (SGD)<\/h3><span id=\"GradientDescent\" class=\"ndg-anchor-alias\" aria-hidden=\"true\"><\/span>\n<p>SGD updates the parameters after every mini-batch. OpenNN supports learning-rate decay, momentum, and Nesterov momentum.<\/p>\n<p><span class=\"nd-math-block\"><math xmlns=\"http:\/\/www.w3.org\/1998\/Math\/MathML\" display=\"block\" aria-label=\"stochastic gradient descent update\"><mrow><msub><mi>\u03b7<\/mi><mi>t<\/mi><\/msub><mo>=<\/mo><mfrac><msub><mi>\u03b7<\/mi><mn>0<\/mn><\/msub><mrow><mn>1<\/mn><mo>+<\/mo><mi>t<\/mi><mi>d<\/mi><\/mrow><\/mfrac><mo>,<\/mo><mspace width=\"1em\"\/><msub><mi>\u03b8<\/mi><mi>t<\/mi><\/msub><mo>=<\/mo><msub><mi>\u03b8<\/mi><mrow><mi>t<\/mi><mo>\u2212<\/mo><mn>1<\/mn><\/mrow><\/msub><mo>\u2212<\/mo><msub><mi>\u03b7<\/mi><mi>t<\/mi><\/msub><msub><mi>g<\/mi><mi>t<\/mi><\/msub><\/mrow><\/math><\/span><\/p>\n<h3 id=\"QuasiNewtonMethod\">Quasi-Newton method (QNM)<\/h3>\n<p>The quasi-Newton method builds a BFGS approximation of the inverse Hessian from successive gradients. An Armijo line search selects the step size.<\/p>\n<p><span class=\"nd-math-block\"><math xmlns=\"http:\/\/www.w3.org\/1998\/Math\/MathML\" display=\"block\" aria-label=\"quasi Newton update\"><mrow><msub><mi>\u03b8<\/mi><mrow><mi>t<\/mi><mo>+<\/mo><mn>1<\/mn><\/mrow><\/msub><mo>=<\/mo><msub><mi>\u03b8<\/mi><mi>t<\/mi><\/msub><mo>\u2212<\/mo><msub><mi>\u03b1<\/mi><mi>t<\/mi><\/msub><msub><mi>H<\/mi><mi>t<\/mi><\/msub><msub><mi>g<\/mi><mi>t<\/mi><\/msub><\/mrow><\/math><\/span><\/p>\n<p>Quasi-Newton uses all training samples in every epoch and runs on CPU. It is the default optimizer for binary and multiclass classification.<\/p>\n<h3 id=\"LevenbergMarquardtAlgorithm\">Levenberg-Marquardt algorithm (LM)<\/h3>\n<p>Levenberg-Marquardt combines the speed of Gauss-Newton with an adaptive damping parameter. It solves the following system for every parameter update:<\/p>\n<p><span class=\"nd-math-block\"><math xmlns=\"http:\/\/www.w3.org\/1998\/Math\/MathML\" display=\"block\" aria-label=\"Levenberg Marquardt update\"><mrow><mo>(<\/mo><msup><mi>J<\/mi><mi>T<\/mi><\/msup><mi>J<\/mi><mo>+<\/mo><mi>\u03bb<\/mi><mi>I<\/mi><mo>)<\/mo><mi>\u0394\u03b8<\/mi><mo>=<\/mo><mo>\u2212<\/mo><msup><mi>J<\/mi><mi>T<\/mi><\/msup><mi>e<\/mi><\/mrow><\/math><\/span><\/p>\n<p>It runs on CPU and is intended for sum-of-squares problems with sequential Dense trainable layers. Softmax outputs, non-Dense trainable layers, and weighted, cross-entropy, or Minkowski errors are not supported.<\/p>\n<h3 id=\"PerformanceConsiderations\">Performance considerations<\/h3>\n<p>Use Levenberg-Marquardt for small Dense regression models and Quasi-Newton for moderate full-batch problems. Use Adam or SGD for GPU training, large data sets, and convolutional, attention, recurrent, image, or text networks.<\/p>\n<p>The <a href=\"https:\/\/www.neuraldesigner.com\/blog\/5_algorithms_to_train_a_neural_network\/\">neural network optimizer guide<\/a> in the <a href=\"https:\/\/www.neuraldesigner.com\/blog\/\">Neural Designer blog<\/a> contains a broader comparison of optimization methods.<\/p>\n<\/div><\/div>\n<div class=\"ndg-step\" id=\"TrainingControl\"><div class=\"ndg-step__no\">3<\/div><div class=\"ndg-step__body\"><h2>Training control and defaults<\/h2>\n<p>The same stopping, validation, and execution controls are shared by the OpenNN optimizers.<\/p>\n<p><img decoding=\"async\" src=\"https:\/\/www.neuraldesigner.com\/images\/training_process.svg\" alt=\"Training process\" width=\"425\" \/><\/p>\n<div class=\"ndg-table-wrap\"><table class=\"ndg-table\"><thead><tr><th>Control<\/th><th>Purpose<\/th><\/tr><\/thead><tbody>\n<tr><td>Batch size<\/td><td>Sets the samples processed together. A value of zero selects the largest batch that fits the available memory.<\/td><\/tr>\n<tr><td>Sample shuffling<\/td><td>Randomizes training batches at every epoch.<\/td><\/tr>\n<tr><td>Loss goal<\/td><td>Stops training when the requested training error is reached.<\/td><\/tr>\n<tr><td>Maximum epochs<\/td><td>Limits the number of complete passes through the training data.<\/td><\/tr>\n<tr><td>Maximum time<\/td><td>Limits the elapsed training time.<\/td><\/tr>\n<tr><td>Validation failures<\/td><td>Stops after the validation error fails to improve a specified number of times.<\/td><\/tr>\n<tr><td>Validation period<\/td><td>Controls how often the validation subset is evaluated.<\/td><\/tr>\n<tr><td>Restore best<\/td><td>Restores the parameters and states from the best validation epoch.<\/td><\/tr>\n<tr><td>Gradient clipping<\/td><td>Limits the gradient norm to stabilize Adam and SGD.<\/td><\/tr>\n<\/tbody><\/table><\/div>\n<h3 id=\"StoppingCriteria\">Stopping criteria<\/h3>\n<p>Training can stop after reaching the loss goal, maximum number of epochs, maximum time, or maximum validation failures. Minimum loss decrease is also available for the full-batch Quasi-Newton and Levenberg-Marquardt algorithms.<\/p>\n<h3 id=\"Validation\">Validation and best model<\/h3>\n<p>Validation samples monitor generalization without updating the parameters. When validation is available, OpenNN stores the best parameters and network states and restores them after training by default.<\/p>\n<p>The testing subset is not used to choose parameters or stop training. It is reserved for the subsequent <a href=\"https:\/\/www.neuraldesigner.com\/learning\/tutorials\/testing-analysis\/\">testing analysis<\/a>.<\/p>\n<h3 id=\"TaskDefaults\">Task defaults<\/h3>\n<p>OpenNN initializes the loss and optimizer from the neural network task:<\/p>\n<div class=\"ndg-table-wrap\"><table class=\"ndg-table\"><thead><tr><th>Task<\/th><th>Default loss<\/th><th>Default optimizer<\/th><\/tr><\/thead><tbody>\n<tr><td>Approximation<\/td><td>Mean squared error<\/td><td>Adam<\/td><\/tr>\n<tr><td>Forecasting<\/td><td>Mean squared error<\/td><td>Adam<\/td><\/tr>\n<tr><td>Auto-association<\/td><td>Mean squared error<\/td><td>Adam<\/td><\/tr>\n<tr><td>Binary classification<\/td><td>Weighted squared error<\/td><td>Quasi-Newton<\/td><\/tr>\n<tr><td>Multiclass classification<\/td><td>Cross-entropy<\/td><td>Quasi-Newton<\/td><\/tr>\n<tr><td>Image classification<\/td><td>Cross-entropy<\/td><td>Adam<\/td><\/tr>\n<tr><td>Text classification<\/td><td>Cross-entropy<\/td><td>Adam<\/td><\/tr>\n<tr><td>Language modeling<\/td><td>3D cross-entropy<\/td><td>Adam<\/td><\/tr>\n<\/tbody><\/table><\/div>\n<h3 id=\"HardwareAndBatches\">Hardware and batches<\/h3>\n<p>Adam and SGD train on CPU or GPU and support CUDA graph execution. OpenNN can size batches from available CPU or GPU memory, shuffle samples, and prefetch batches during training. Quasi-Newton and Levenberg-Marquardt are full-batch CPU algorithms.<\/p>\n<h3 id=\"TrainingResults\">Training results<\/h3>\n<p>The training result stores the training and validation histories, stopping condition, elapsed and measured training time, final loss, and the epoch restored as the best model.<\/p>\n<\/div><\/div>\n<div class=\"ndg-nav\"><a href=\"https:\/\/www.neuraldesigner.com\/learning\/tutorials\/neural-network\/\">\u21d0 Neural Network<\/a><a href=\"https:\/\/www.neuraldesigner.com\/learning\/tutorials\/model-selection\/\">Model Selection \u21d2<\/a><\/div>\n<\/div><\/div>\n","protected":false},"author":122,"featured_media":1428,"template":"","categories":[30],"tags":[36],"class_list":["post-3539","learning","type-learning","status-publish","has-post-thumbnail","hentry","category-tutorials","tag-tutorials"],"acf":[],"yoast_head":"<!-- This site is optimized with the Yoast SEO plugin v26.4 - https:\/\/yoast.com\/wordpress\/plugins\/seo\/ -->\n<title>Machine learning: Training strategy - tutorial<\/title>\n<meta name=\"description\" content=\"This tutorial shows the main training strategies used by neural networks to learn.\" \/>\n<meta name=\"robots\" content=\"index, follow, max-snippet:-1, max-image-preview:large, max-video-preview:-1\" \/>\n<link rel=\"canonical\" href=\"https:\/\/www.neuraldesigner.com\/learning\/tutorials\/training-strategy\/\" \/>\n<meta property=\"og:locale\" content=\"en_US\" \/>\n<meta property=\"og:type\" content=\"article\" \/>\n<meta property=\"og:title\" content=\"Neural networks tutorial: Training strategy\" \/>\n<meta property=\"og:description\" content=\"This tutorial shows the main training strategies used by neural networks to learn.\" \/>\n<meta property=\"og:url\" content=\"https:\/\/www.neuraldesigner.com\/learning\/tutorials\/training-strategy\/\" \/>\n<meta property=\"og:site_name\" content=\"Neural Designer\" \/>\n<meta property=\"article:modified_time\" content=\"2026-08-26T08:53:57+00:00\" \/>\n<meta name=\"twitter:card\" content=\"summary_large_image\" \/>\n<meta name=\"twitter:title\" content=\"Neural networks tutorial: Training strategy\" \/>\n<meta name=\"twitter:description\" content=\"This tutorial shows the main training strategies used by neural networks to learn.\" \/>\n<meta name=\"twitter:site\" content=\"@NeuralDesigner\" \/>\n<meta name=\"twitter:label1\" content=\"Est. reading time\" \/>\n\t<meta name=\"twitter:data1\" content=\"7 minutes\" \/>\n<script type=\"application\/ld+json\" class=\"yoast-schema-graph\">{\"@context\":\"https:\/\/schema.org\",\"@graph\":[{\"@type\":\"WebPage\",\"@id\":\"https:\/\/www.neuraldesigner.com\/learning\/tutorials\/training-strategy\/\",\"url\":\"https:\/\/www.neuraldesigner.com\/learning\/tutorials\/training-strategy\/\",\"name\":\"Machine learning: Training strategy - tutorial\",\"isPartOf\":{\"@id\":\"https:\/\/www.neuraldesigner.com\/#website\"},\"primaryImageOfPage\":{\"@id\":\"https:\/\/www.neuraldesigner.com\/learning\/tutorials\/training-strategy\/#primaryimage\"},\"image\":{\"@id\":\"https:\/\/www.neuraldesigner.com\/learning\/tutorials\/training-strategy\/#primaryimage\"},\"thumbnailUrl\":\"https:\/\/www.neuraldesigner.com\/wp-content\/uploads\/2023\/06\/training_process.svg\",\"datePublished\":\"2025-11-25T10:12:57+00:00\",\"dateModified\":\"2026-08-26T08:53:57+00:00\",\"description\":\"This tutorial shows the main training strategies used by neural networks to learn.\",\"breadcrumb\":{\"@id\":\"https:\/\/www.neuraldesigner.com\/learning\/tutorials\/training-strategy\/#breadcrumb\"},\"inLanguage\":\"en-US\",\"potentialAction\":[{\"@type\":\"ReadAction\",\"target\":[\"https:\/\/www.neuraldesigner.com\/learning\/tutorials\/training-strategy\/\"]}]},{\"@type\":\"ImageObject\",\"inLanguage\":\"en-US\",\"@id\":\"https:\/\/www.neuraldesigner.com\/learning\/tutorials\/training-strategy\/#primaryimage\",\"url\":\"https:\/\/www.neuraldesigner.com\/wp-content\/uploads\/2023\/06\/training_process.svg\",\"contentUrl\":\"https:\/\/www.neuraldesigner.com\/wp-content\/uploads\/2023\/06\/training_process.svg\",\"width\":13341,\"height\":8113,\"caption\":\"Neural network training process diagram\"},{\"@type\":\"BreadcrumbList\",\"@id\":\"https:\/\/www.neuraldesigner.com\/learning\/tutorials\/training-strategy\/#breadcrumb\",\"itemListElement\":[{\"@type\":\"ListItem\",\"position\":1,\"name\":\"Home\",\"item\":\"https:\/\/www.neuraldesigner.com\/\"},{\"@type\":\"ListItem\",\"position\":2,\"name\":\"Learning\",\"item\":\"https:\/\/www.neuraldesigner.com\/learning\/\"},{\"@type\":\"ListItem\",\"position\":3,\"name\":\"Machine learning: Training strategy &#8211; tutorial\"}]},{\"@type\":\"WebSite\",\"@id\":\"https:\/\/www.neuraldesigner.com\/#website\",\"url\":\"https:\/\/www.neuraldesigner.com\/\",\"name\":\"Neural Designer\",\"description\":\"Explainable AI Platform\",\"publisher\":{\"@id\":\"https:\/\/www.neuraldesigner.com\/#organization\"},\"potentialAction\":[{\"@type\":\"SearchAction\",\"target\":{\"@type\":\"EntryPoint\",\"urlTemplate\":\"https:\/\/www.neuraldesigner.com\/?s={search_term_string}\"},\"query-input\":{\"@type\":\"PropertyValueSpecification\",\"valueRequired\":true,\"valueName\":\"search_term_string\"}}],\"inLanguage\":\"en-US\"},{\"@type\":\"Organization\",\"@id\":\"https:\/\/www.neuraldesigner.com\/#organization\",\"name\":\"Neural Designer\",\"url\":\"https:\/\/www.neuraldesigner.com\/\",\"logo\":{\"@type\":\"ImageObject\",\"inLanguage\":\"en-US\",\"@id\":\"https:\/\/www.neuraldesigner.com\/#\/schema\/logo\/image\/\",\"url\":\"https:\/\/www.neuraldesigner.com\/wp-content\/uploads\/2023\/05\/logo-neural-1.png\",\"contentUrl\":\"https:\/\/www.neuraldesigner.com\/wp-content\/uploads\/2023\/05\/logo-neural-1.png\",\"width\":1024,\"height\":223,\"caption\":\"Neural Designer\"},\"image\":{\"@id\":\"https:\/\/www.neuraldesigner.com\/#\/schema\/logo\/image\/\"},\"sameAs\":[\"https:\/\/x.com\/NeuralDesigner\",\"https:\/\/es.linkedin.com\/showcase\/neuraldesigner\/\"]}]}<\/script>\n<!-- \/ Yoast SEO plugin. -->","yoast_head_json":{"title":"Machine learning: Training strategy - tutorial","description":"This tutorial shows the main training strategies used by neural networks to learn.","robots":{"index":"index","follow":"follow","max-snippet":"max-snippet:-1","max-image-preview":"max-image-preview:large","max-video-preview":"max-video-preview:-1"},"canonical":"https:\/\/www.neuraldesigner.com\/learning\/tutorials\/training-strategy\/","og_locale":"en_US","og_type":"article","og_title":"Neural networks tutorial: Training strategy","og_description":"This tutorial shows the main training strategies used by neural networks to learn.","og_url":"https:\/\/www.neuraldesigner.com\/learning\/tutorials\/training-strategy\/","og_site_name":"Neural Designer","article_modified_time":"2026-08-26T08:53:57+00:00","twitter_card":"summary_large_image","twitter_title":"Neural networks tutorial: Training strategy","twitter_description":"This tutorial shows the main training strategies used by neural networks to learn.","twitter_site":"@NeuralDesigner","twitter_misc":{"Est. reading time":"7 minutes"},"schema":{"@context":"https:\/\/schema.org","@graph":[{"@type":"WebPage","@id":"https:\/\/www.neuraldesigner.com\/learning\/tutorials\/training-strategy\/","url":"https:\/\/www.neuraldesigner.com\/learning\/tutorials\/training-strategy\/","name":"Machine learning: Training strategy - tutorial","isPartOf":{"@id":"https:\/\/www.neuraldesigner.com\/#website"},"primaryImageOfPage":{"@id":"https:\/\/www.neuraldesigner.com\/learning\/tutorials\/training-strategy\/#primaryimage"},"image":{"@id":"https:\/\/www.neuraldesigner.com\/learning\/tutorials\/training-strategy\/#primaryimage"},"thumbnailUrl":"https:\/\/www.neuraldesigner.com\/wp-content\/uploads\/2023\/06\/training_process.svg","datePublished":"2025-11-25T10:12:57+00:00","dateModified":"2026-08-26T08:53:57+00:00","description":"This tutorial shows the main training strategies used by neural networks to learn.","breadcrumb":{"@id":"https:\/\/www.neuraldesigner.com\/learning\/tutorials\/training-strategy\/#breadcrumb"},"inLanguage":"en-US","potentialAction":[{"@type":"ReadAction","target":["https:\/\/www.neuraldesigner.com\/learning\/tutorials\/training-strategy\/"]}]},{"@type":"ImageObject","inLanguage":"en-US","@id":"https:\/\/www.neuraldesigner.com\/learning\/tutorials\/training-strategy\/#primaryimage","url":"https:\/\/www.neuraldesigner.com\/wp-content\/uploads\/2023\/06\/training_process.svg","contentUrl":"https:\/\/www.neuraldesigner.com\/wp-content\/uploads\/2023\/06\/training_process.svg","width":13341,"height":8113,"caption":"Neural network training process diagram"},{"@type":"BreadcrumbList","@id":"https:\/\/www.neuraldesigner.com\/learning\/tutorials\/training-strategy\/#breadcrumb","itemListElement":[{"@type":"ListItem","position":1,"name":"Home","item":"https:\/\/www.neuraldesigner.com\/"},{"@type":"ListItem","position":2,"name":"Learning","item":"https:\/\/www.neuraldesigner.com\/learning\/"},{"@type":"ListItem","position":3,"name":"Machine learning: Training strategy &#8211; tutorial"}]},{"@type":"WebSite","@id":"https:\/\/www.neuraldesigner.com\/#website","url":"https:\/\/www.neuraldesigner.com\/","name":"Neural Designer","description":"Explainable AI Platform","publisher":{"@id":"https:\/\/www.neuraldesigner.com\/#organization"},"potentialAction":[{"@type":"SearchAction","target":{"@type":"EntryPoint","urlTemplate":"https:\/\/www.neuraldesigner.com\/?s={search_term_string}"},"query-input":{"@type":"PropertyValueSpecification","valueRequired":true,"valueName":"search_term_string"}}],"inLanguage":"en-US"},{"@type":"Organization","@id":"https:\/\/www.neuraldesigner.com\/#organization","name":"Neural Designer","url":"https:\/\/www.neuraldesigner.com\/","logo":{"@type":"ImageObject","inLanguage":"en-US","@id":"https:\/\/www.neuraldesigner.com\/#\/schema\/logo\/image\/","url":"https:\/\/www.neuraldesigner.com\/wp-content\/uploads\/2023\/05\/logo-neural-1.png","contentUrl":"https:\/\/www.neuraldesigner.com\/wp-content\/uploads\/2023\/05\/logo-neural-1.png","width":1024,"height":223,"caption":"Neural Designer"},"image":{"@id":"https:\/\/www.neuraldesigner.com\/#\/schema\/logo\/image\/"},"sameAs":["https:\/\/x.com\/NeuralDesigner","https:\/\/es.linkedin.com\/showcase\/neuraldesigner\/"]}]}},"_links":{"self":[{"href":"https:\/\/www.neuraldesigner.com\/api\/wp\/v2\/learning\/3539","targetHints":{"allow":["GET"]}}],"collection":[{"href":"https:\/\/www.neuraldesigner.com\/api\/wp\/v2\/learning"}],"about":[{"href":"https:\/\/www.neuraldesigner.com\/api\/wp\/v2\/types\/learning"}],"author":[{"embeddable":true,"href":"https:\/\/www.neuraldesigner.com\/api\/wp\/v2\/users\/122"}],"version-history":[{"count":9,"href":"https:\/\/www.neuraldesigner.com\/api\/wp\/v2\/learning\/3539\/revisions"}],"predecessor-version":[{"id":23722,"href":"https:\/\/www.neuraldesigner.com\/api\/wp\/v2\/learning\/3539\/revisions\/23722"}],"wp:featuredmedia":[{"embeddable":true,"href":"https:\/\/www.neuraldesigner.com\/api\/wp\/v2\/media\/1428"}],"wp:attachment":[{"href":"https:\/\/www.neuraldesigner.com\/api\/wp\/v2\/media?parent=3539"}],"wp:term":[{"taxonomy":"category","embeddable":true,"href":"https:\/\/www.neuraldesigner.com\/api\/wp\/v2\/categories?post=3539"},{"taxonomy":"post_tag","embeddable":true,"href":"https:\/\/www.neuraldesigner.com\/api\/wp\/v2\/tags?post=3539"}],"curies":[{"name":"wp","href":"https:\/\/api.w.org\/{rel}","templated":true}]}}