\begin{thebibliography}{10}
\expandafter\ifx\csname url\endcsname\relax
  \def\url#1{\texttt{#1}}\fi
\expandafter\ifx\csname urlprefix\endcsname\relax\def\urlprefix{URL }\fi
\expandafter\ifx\csname href\endcsname\relax
  \def\href#1#2{#2} \def\path#1{#1}\fi

\bibitem{robbins1951stochastic}
H.~Robbins, S.~Monro, A stochastic approximation method, The annals of
  mathematical statistics (1951) 400--407.

\bibitem{cauchy1847methode}
A.~Cauchy, M{'e}thode g{'e}n{'e}rale pour la r{'e}solution des systemes
  d’{'e}quations simultan{'e}es, Comp. Rend. Sci. Paris 25~(1847) (1847)
  536--538.

\bibitem{marchuk1968some}
G.~I. Marchuk, Some application of splitting-up methods to the solution of
  mathematical physics problems, Aplikace matematiky 13~(2) (1968) 103--132.

\bibitem{strang1968construction}
G.~Strang, On the construction and comparison of difference schemes, SIAM
  Journal on Numerical Analysis 5~(3) (1968) 506--517.

\bibitem{dormand1980family}
J.~R. Dormand, P.~J. Prince, A family of embedded runge-kutta formulae, Journal
  of computational and applied mathematics 6~(1) (1980) 19--26.

\bibitem{shampine1986some}
L.~F. Shampine, Some practical runge-kutta formulas, Mathematics of computation
  46~(173) (1986) 135--150.

\bibitem{2020SciPy-NMeth}
P.~{Virtanen}, R.~{Gommers}, T.~E. {Oliphant}, M.~{Haberland}, T.~{Reddy},
  D.~{Cournapeau}, E.~{Burovski}, P.~{Peterson}, W.~{Weckesser}, J.~{Bright},
  S.~J. {van der Walt}, M.~{Brett}, J.~{Wilson}, K.~{Jarrod Millman},
  N.~{Mayorov}, A.~R.~J. {Nelson}, E.~{Jones}, R.~{Kern}, E.~{Larson},
  C.~{Carey}, {\.I}.~{Polat}, Y.~{Feng}, E.~W. {Moore}, J.~{Vand erPlas},
  D.~{Laxalde}, J.~{Perktold}, R.~{Cimrman}, I.~{Henriksen}, E.~A. {Quintero},
  C.~R. {Harris}, A.~M. {Archibald}, A.~H. {Ribeiro}, F.~{Pedregosa}, P.~{van
  Mulbregt}, S.~.~. {Contributors}, {SciPy 1.0: Fundamental Algorithms for
  Scientific Computing in Python}, Nature Methods 17 (2020) 261--272.
\newblock \href {https://doi.org/https://doi.org/10.1038/s41592-019-0686-2}
  {\path{doi:https://doi.org/10.1038/s41592-019-0686-2}}.

\bibitem{kaczmarz1937method}
S.~Kaczmarz., Angenäherte auflösung von systemen linearer gleichungen, Bull.
  Internat. Acad. Polon.Sci. Lettres A (1937) 335--357.

\bibitem{strohmer2009randomized}
T.~Strohmer, R.~Vershynin, A randomized kaczmarz algorithm with exponential
  convergence, Journal of Fourier Analysis and Applications 15~(2) (2009) 262.

\bibitem{gower2015randomized}
R.~M. Gower, P.~Richt{\'a}rik, Randomized iterative methods for linear systems,
  SIAM Journal on Matrix Analysis and Applications 36~(4) (2015) 1660--1690.

\bibitem{needell2014stochastic}
D.~Needell, R.~Ward, N.~Srebro, Stochastic gradient descent, weighted sampling,
  and the randomized kaczmarz algorithm, in: Advances in neural information
  processing systems, 2014, pp. 1017--1025.

\bibitem{hansen2018air}
P.~C. Hansen, J.~S. J{\o}rgensen, Air tools ii: algebraic iterative
  reconstruction methods, improved implementation, Numerical Algorithms 79~(1)
  (2018) 107--137.

\bibitem{lecun1998gradient}
Y.~LeCun, L.~Bottou, Y.~Bengio, P.~Haffner, Gradient-based learning applied to
  document recognition, Proceedings of the IEEE 86~(11) (1998) 2278--2324.

\bibitem{xiao2017fashion}
H.~Xiao, K.~Rasul, R.~Vollgraf, Fashion-mnist: a novel image dataset for
  benchmarking machine learning algorithms, arXiv preprint arXiv:1708.07747
  (2017).

\bibitem{su2014differential}
W.~Su, S.~Boyd, E.~Candes, A differential equation for modeling nesterov’s
  accelerated gradient method: Theory and insights, in: Advances in Neural
  Information Processing Systems, 2014, pp. 2510--2518.

\bibitem{nesterov1983method}
Y.~E. Nesterov, A method of solving a convex programming problem with
  convergence rate $o(k^2)$, in: Doklady Akademii Nauk, Vol. 269, Russian
  Academy of Sciences, 1983, pp. 543--547.

\bibitem{wibisono2016variational}
A.~Wibisono, A.~C. Wilson, M.~I. Jordan, A variational perspective on
  accelerated methods in optimization, proceedings of the National Academy of
  Sciences 113~(47) (2016) E7351--E7358.

\bibitem{helmke2012optimization}
U.~Helmke, J.~B. Moore, Optimization and dynamical systems, Springer Science \&
  Business Media, 2012.

\bibitem{evtushenko1994stable}
Y.~G. Evtushenko, V.~G. Zhadan, Stable barrier-projection and barrier-newton
  methods in linear programming, Computational Optimization and Applications
  3~(4) (1994) 289--303.

\bibitem{blanes2024splitting}
S.~Blanes, F.~Casas, A.~Murua, Splitting and composition methods with embedded
  error estimators, Acta Numerica 33 (2024) 1--161.
\newblock \href {https://doi.org/10.1017/S0962492924000011}
  {\path{doi:10.1017/S0962492924000011}}.

\bibitem{huang2022dosnet}
W.~Huang, W.~Yin, A deeper look into deep operator splitting networks, Journal
  of Computational Physics 498 (2024) 112641.

\bibitem{tibshirani1996lasso}
R.~Tibshirani, Regression shrinkage and selection via the lasso, Journal of the
  Royal Statistical Society: Series B (Methodological) 58~(1) (1996) 267--288.

\bibitem{beck2009fast}
A.~Beck, M.~Teboulle, A fast iterative shrinkage-thresholding algorithm for
  linear inverse problems, SIAM journal on imaging sciences 2~(1) (2009)
  183--202.

\bibitem{parikh2014proximal}
N.~Parikh, S.~Boyd, Proximal algorithms, Foundations and Trends in Optimization
  1~(3) (2014) 127--239.

\bibitem{sheng1994global}
Q.~Sheng, Global error estimates for exponential splitting, IMA Journal of
  Numerical Analysis 14~(1) (1994) 27--56.

\end{thebibliography}
