% Survey cutoff: 2026-10-03. All entries use @misc to avoid inventing unverified venue metadata.
% This is a working bibliography; verify complete author lists and publication metadata for submission.

@misc{OP001,
  author = {Diederik P. Kingma and Jimmy Ba},
  title = {{Adam: A Method for Stochastic Optimization}},
  year = {2014},
  url = {https://arxiv.org/abs/1412.6980},
  note = {OP001: ICLR 2015（arXiv初出2014）},
  urldate = {2026-10-03},
  eprint = {1412.6980},
  archivePrefix = {arXiv}
}

@misc{OP002,
  author = {Noam Shazeer and Mitchell Stern},
  title = {{Adafactor: Adaptive Learning Rates with Sublinear Memory Cost}},
  year = {2018},
  url = {https://arxiv.org/abs/1804.04235},
  note = {OP002: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {1804.04235},
  archivePrefix = {arXiv}
}

@misc{OP003,
  author = {Xiangning Chen and Chen Liang and Da Huang and others},
  title = {{Symbolic Discovery of Optimization Algorithms}},
  year = {2023},
  url = {https://arxiv.org/abs/2302.06675},
  note = {OP003: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2302.06675},
  archivePrefix = {arXiv}
}

@misc{OP004,
  author = {Dami Choi and Christopher J. Shallue and Zachary Nado and others},
  title = {{On Empirical Comparisons of Optimizers for Deep Learning}},
  year = {2019},
  url = {https://arxiv.org/abs/1910.05446},
  note = {OP004: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {1910.05446},
  archivePrefix = {arXiv}
}

@misc{OP005,
  author = {George E. Dahl and Frank Schneider and Zachary Nado and others},
  title = {{Benchmarking Neural Network Training Algorithms}},
  year = {2023},
  url = {https://arxiv.org/abs/2306.07179},
  note = {OP005: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2306.07179},
  archivePrefix = {arXiv}
}

@misc{OP006,
  author = {Vineet Gupta and Tomer Koren and Yoram Singer},
  title = {{Shampoo: Preconditioned Stochastic Tensor Optimization}},
  year = {2018},
  url = {https://arxiv.org/abs/1802.09568},
  note = {OP006: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {1802.09568},
  archivePrefix = {arXiv}
}

@misc{OP007,
  author = {Rohan Anil and Vineet Gupta and Tomer Koren and Kevin Regan and Yoram Singer},
  title = {{Scalable Second Order Optimization for Deep Learning}},
  year = {2020},
  url = {https://arxiv.org/abs/2002.09018},
  note = {OP007: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2002.09018},
  archivePrefix = {arXiv}
}

@misc{OP008,
  author = {Nikhil Vyas and Depen Morwani and Rosie Zhao and others},
  title = {{SOAP: Improving and Stabilizing Shampoo using Adam}},
  year = {2024},
  url = {https://arxiv.org/abs/2409.11321},
  note = {OP008: ICLR 2025（arXiv初出2024、会議版題名末尾にfor Language Modeling）},
  urldate = {2026-10-03},
  eprint = {2409.11321},
  archivePrefix = {arXiv}
}

@misc{OP009,
  author = {Jingyuan Liu and Jianlin Su and Xingcheng Yao and others},
  title = {{Muon is Scalable for LLM Training}},
  year = {2025},
  url = {https://arxiv.org/abs/2502.16982},
  note = {OP009: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2502.16982},
  archivePrefix = {arXiv}
}

@misc{OP010,
  author = {Essential AI : Ishaan Shah and Anthony M. Polloreno and Karl Stratos and others},
  title = {{Practical Efficiency of Muon for Pretraining}},
  year = {2025},
  url = {https://arxiv.org/abs/2505.02222},
  note = {OP010: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2505.02222},
  archivePrefix = {arXiv}
}

@misc{OP011,
  author = {Naoki Sato and Hiroki Naganuma and Hideaki Iiduka},
  title = {{Convergence Bound and Critical Batch Size of Muon Optimizer}},
  year = {2025},
  url = {https://arxiv.org/abs/2507.01598},
  note = {OP011: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2507.01598},
  archivePrefix = {arXiv}
}

@misc{OP012,
  author = {Thang Do and Steffen Dereich and Arnulf Jentzen},
  title = {{On MUON optimization: From non-convergence to an error analysis with Polar Express and the Newton-Schulz polynomial from implementations}},
  year = {2026},
  url = {https://arxiv.org/abs/2608.04607},
  note = {OP012: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2608.04607},
  archivePrefix = {arXiv}
}

@misc{OP013,
  author = {Arthur C. B. de Oliveira and Dhruv D. Jatkar and Guilherme S. Vicinansa and Eduardo D. Sontag},
  title = {{Convergence guarantees for Muon: New parameter regimes and generalizations}},
  year = {2026},
  url = {https://arxiv.org/abs/2609.30546},
  note = {OP013: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2609.30546},
  archivePrefix = {arXiv}
}

@misc{OP014,
  author = {Mikail Khona and Aditya Vavre and Boxiang Wang and others},
  title = {{SOAP, Muon, and Beyond: Pushing LLM Pretraining Scales}},
  year = {2026},
  url = {https://arxiv.org/abs/2607.20548},
  note = {OP014: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2607.20548},
  archivePrefix = {arXiv}
}

@misc{OP015,
  author = {James Martens and Roger Grosse},
  title = {{Optimizing Neural Networks with Kronecker-factored Approximate Curvature}},
  year = {2015},
  url = {https://arxiv.org/abs/1503.05671},
  note = {OP015: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {1503.05671},
  archivePrefix = {arXiv}
}

@misc{OP016,
  author = {Roger Grosse and James Martens},
  title = {{A Kronecker-factored approximate Fisher matrix for convolution layers}},
  year = {2016},
  url = {https://arxiv.org/abs/1602.01407},
  note = {OP016: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {1602.01407},
  archivePrefix = {arXiv}
}

@misc{OP017,
  author = {Runa Eschenhagen and Alexander Immer and Richard E. Turner and Frank Schneider and Philipp Hennig},
  title = {{Kronecker-Factored Approximate Curvature for Modern Neural Network Architectures}},
  year = {2023},
  url = {https://arxiv.org/abs/2311.00636},
  note = {OP017: NeurIPS 2023},
  urldate = {2026-10-03},
  eprint = {2311.00636},
  archivePrefix = {arXiv}
}

@misc{OP018,
  author = {Jeremy Bernstein and Laker Newhouse},
  title = {{Old Optimizer, New Norm: An Anthology}},
  year = {2024},
  url = {https://arxiv.org/abs/2409.20325},
  note = {OP018: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2409.20325},
  archivePrefix = {arXiv}
}

@misc{OP019,
  author = {Noah Amsel and David Persson and Christopher Musco and Robert M. Gower},
  title = {{The Polar Express: Optimal Matrix Sign Methods and Their Application to the Muon Algorithm}},
  year = {2025},
  url = {https://arxiv.org/abs/2505.16932},
  note = {OP019: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2505.16932},
  archivePrefix = {arXiv}
}

@misc{OP020,
  author = {Zichong Li and Liming Liu and Chen Liang and Weizhu Chen and Tuo Zhao},
  title = {{NorMuon: Making Muon more efficient and scalable}},
  year = {2025},
  url = {https://arxiv.org/abs/2510.05491},
  note = {OP020: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2510.05491},
  archivePrefix = {arXiv}
}

@misc{OP021,
  author = {Priya Goyal and Piotr Dollár and Ross Girshick and others},
  title = {{Accurate, Large Minibatch SGD: Training ImageNet in 1 Hour}},
  year = {2017},
  url = {https://arxiv.org/abs/1706.02677},
  note = {OP021: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {1706.02677},
  archivePrefix = {arXiv}
}

@misc{OP022,
  author = {Sam McCandlish and Jared Kaplan and Dario Amodei and OpenAI Dota Team},
  title = {{An Empirical Model of Large-Batch Training}},
  year = {2018},
  url = {https://arxiv.org/abs/1812.06162},
  note = {OP022: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {1812.06162},
  archivePrefix = {arXiv}
}

@misc{OP023,
  author = {Guodong Zhang and Lala Li and Zachary Nado and others},
  title = {{Which Algorithmic Choices Matter at Which Batch Sizes? Insights From a Noisy Quadratic Model}},
  year = {2019},
  url = {https://arxiv.org/abs/1907.04164},
  note = {OP023: NeurIPS 2019},
  urldate = {2026-10-03},
  eprint = {1907.04164},
  archivePrefix = {arXiv}
}

@misc{OP024,
  author = {Yang You and Igor Gitman and Boris Ginsburg},
  title = {{Large Batch Training of Convolutional Networks}},
  year = {2017},
  url = {https://arxiv.org/abs/1708.03888},
  note = {OP024: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {1708.03888},
  archivePrefix = {arXiv}
}

@misc{OP025,
  author = {Yang You and Jing Li and Sashank Reddi and others},
  title = {{Large Batch Optimization for Deep Learning: Training BERT in 76 minutes}},
  year = {2019},
  url = {https://arxiv.org/abs/1904.00962},
  note = {OP025: ICLR 2020（arXiv初出2019）},
  urldate = {2026-10-03},
  eprint = {1904.00962},
  archivePrefix = {arXiv}
}

@misc{OP026,
  author = {Zachary Nado and Justin M. Gilmer and Christopher J. Shallue and Rohan Anil and George E. Dahl},
  title = {{A Large Batch Optimizer Reality Check: Traditional, Generic Optimizers Suffice Across Batch Sizes}},
  year = {2021},
  url = {https://arxiv.org/abs/2102.06356},
  note = {OP026: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2102.06356},
  archivePrefix = {arXiv}
}

@misc{OP027,
  author = {Aaron Defazio and Xingyu Alice Yang and Harsh Mehta and others},
  title = {{The Road Less Scheduled}},
  year = {2024},
  url = {https://arxiv.org/abs/2405.15682},
  note = {OP027: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2405.15682},
  archivePrefix = {arXiv}
}

@misc{OP028,
  author = {Fabian Schaipp and Alexander Hägele and Adrien Taylor and Umut Simsekli and Francis Bach},
  title = {{The Surprising Agreement Between Convex Optimization Theory and Learning-Rate Scheduling for Large Model Training}},
  year = {2025},
  url = {https://arxiv.org/abs/2501.18965},
  note = {OP028: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2501.18965},
  archivePrefix = {arXiv}
}

@misc{OP029,
  author = {Kaiyue Wen and Zhiyuan Li and Jason Wang and others},
  title = {{Understanding Warmup-Stable-Decay Learning Rates: A River Valley Loss Landscape Perspective}},
  year = {2024},
  url = {https://arxiv.org/abs/2410.05192},
  note = {OP029: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2410.05192},
  archivePrefix = {arXiv}
}

@misc{OP030,
  author = {Sebastian U. Stich},
  title = {{Local SGD Converges Fast and Communicates Little}},
  year = {2018},
  url = {https://arxiv.org/abs/1805.09767},
  note = {OP030: ICLR 2019（arXiv初出2018）},
  urldate = {2026-10-03},
  eprint = {1805.09767},
  archivePrefix = {arXiv}
}

@misc{OP031,
  author = {Tao Lin and Sebastian U. Stich and Kumar Kshitij Patel and Martin Jaggi},
  title = {{Don't Use Large Mini-Batches, Use Local SGD}},
  year = {2018},
  url = {https://arxiv.org/abs/1808.07217},
  note = {OP031: ICLR 2020（arXiv初出2018）},
  urldate = {2026-10-03},
  eprint = {1808.07217},
  archivePrefix = {arXiv}
}

@misc{OP032,
  author = {Jianyu Wang and Vinayak Tantia and Nicolas Ballas and Michael Rabbat},
  title = {{SlowMo: Improving Communication-Efficient Distributed SGD with Slow Momentum}},
  year = {2019},
  url = {https://arxiv.org/abs/1910.00643},
  note = {OP032: ICLR 2020（arXiv初出2019）},
  urldate = {2026-10-03},
  eprint = {1910.00643},
  archivePrefix = {arXiv}
}

@misc{OP033,
  author = {Arthur Douillard and Qixuan Feng and Andrei A. Rusu and others},
  title = {{DiLoCo: Distributed Low-Communication Training of Language Models}},
  year = {2023},
  url = {https://arxiv.org/abs/2311.08105},
  note = {OP033: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2311.08105},
  archivePrefix = {arXiv}
}

@misc{OP034,
  author = {Bo Liu and Rachita Chhaparia and Arthur Douillard and others},
  title = {{Asynchronous Local-SGD Training for Language Modeling}},
  year = {2024},
  url = {https://arxiv.org/abs/2401.09135},
  note = {OP034: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2401.09135},
  archivePrefix = {arXiv}
}

@misc{OP035,
  author = {Arthur Douillard and Keith Rush and Yani Donchev and others},
  title = {{Decoupled DiLoCo for Resilient Distributed Pre-training}},
  year = {2026},
  url = {https://arxiv.org/abs/2604.21428},
  note = {OP035: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2604.21428},
  archivePrefix = {arXiv}
}

@misc{OP036,
  author = {H. Brendan McMahan and Eider Moore and Daniel Ramage and Seth Hampson and Blaise Agüera y Arcas},
  title = {{Communication-Efficient Learning of Deep Networks from Decentralized Data}},
  year = {2016},
  url = {https://arxiv.org/abs/1602.05629},
  note = {OP036: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {1602.05629},
  archivePrefix = {arXiv}
}

@misc{OP037,
  author = {Sashank Reddi and Zachary Charles and Manzil Zaheer and others},
  title = {{Adaptive Federated Optimization}},
  year = {2020},
  url = {https://arxiv.org/abs/2003.00295},
  note = {OP037: ICLR 2021（arXiv初出2020）},
  urldate = {2026-10-03},
  eprint = {2003.00295},
  archivePrefix = {arXiv}
}

@misc{OP038,
  author = {Sai Praneeth Karimireddy and Satyen Kale and Mehryar Mohri and others},
  title = {{SCAFFOLD: Stochastic Controlled Averaging for Federated Learning}},
  year = {2019},
  url = {https://arxiv.org/abs/1910.06378},
  note = {OP038: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {1910.06378},
  archivePrefix = {arXiv}
}

@misc{OP039,
  author = {Ilya Loshchilov and Frank Hutter},
  title = {{SGDR: Stochastic Gradient Descent with Warm Restarts}},
  year = {2016},
  url = {https://arxiv.org/abs/1608.03983},
  note = {OP039: ICLR 2017（arXiv初出2016）},
  urldate = {2026-10-03},
  eprint = {1608.03983},
  archivePrefix = {arXiv}
}

@misc{OP040,
  author = {Samuel L. Smith and Pieter-Jan Kindermans and Chris Ying and Quoc V. Le},
  title = {{Don't Decay the Learning Rate, Increase the Batch Size}},
  year = {2017},
  url = {https://arxiv.org/abs/1711.00489},
  note = {OP040: ICLR 2018（arXiv初出2017）},
  urldate = {2026-10-03},
  eprint = {1711.00489},
  archivePrefix = {arXiv}
}

@misc{OP041,
  author = {Binghui Li and Zilin Wang and Fengling Chen and others},
  title = {{Optimal Learning Rate Schedules under Functional Scaling Laws: Power Decay and Warmup-Stable-Decay}},
  year = {2026},
  url = {https://arxiv.org/abs/2602.06797},
  note = {OP041: COLT 2026（arXiv採択注記を確認）},
  urldate = {2026-10-03},
  eprint = {2602.06797},
  archivePrefix = {arXiv}
}

@misc{OP042,
  author = {Jiseok Chae and Donghwan Kim},
  title = {{Understanding Schedule-Free Methods in Nonconvex Optimization: Rate Guarantees and Escaping Saddles}},
  year = {2026},
  url = {https://arxiv.org/abs/2607.09167},
  note = {OP042: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2607.09167},
  archivePrefix = {arXiv}
}

@misc{OP043,
  author = {Jianhao Ma and Yuxin Chen},
  title = {{WSqD: A Horizon-Free Learning Rate Schedule for Large Model Training}},
  year = {2026},
  url = {https://arxiv.org/abs/2607.10959},
  note = {OP043: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2607.10959},
  archivePrefix = {arXiv}
}

@misc{OP044,
  author = {Kwangjun Ahn and Byron Xu and Natalie Abreu and others},
  title = {{Dion: Distributed Orthonormalized Updates}},
  year = {2025},
  url = {https://arxiv.org/abs/2504.05295},
  note = {OP044: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2504.05295},
  archivePrefix = {arXiv}
}

@misc{OP045,
  author = {Pierre Foret and Ariel Kleiner and Hossein Mobahi and Behnam Neyshabur},
  title = {{Sharpness-Aware Minimization for Efficiently Improving Generalization}},
  year = {2020},
  url = {https://arxiv.org/abs/2010.01412},
  note = {OP045: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2010.01412},
  archivePrefix = {arXiv}
}

@misc{OP046,
  author = {Maksym Andriushchenko and Dara Bahri and Hossein Mobahi and Nicolas Flammarion},
  title = {{Sharpness-Aware Minimization Leads to Low-Rank Features}},
  year = {2023},
  url = {https://arxiv.org/abs/2305.16292},
  note = {OP046: NeurIPS 2023},
  urldate = {2026-10-03},
  eprint = {2305.16292},
  archivePrefix = {arXiv}
}

@misc{OP047,
  author = {Kaiyue Wen and Tengyu Ma and Zhiyuan Li},
  title = {{How Does Sharpness-Aware Minimization Minimize Sharpness?}},
  year = {2022},
  url = {https://arxiv.org/abs/2211.05729},
  note = {OP047: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2211.05729},
  archivePrefix = {arXiv}
}

@misc{OP048,
  author = {Ilya Loshchilov and Frank Hutter},
  title = {{Decoupled Weight Decay Regularization}},
  year = {2017},
  url = {https://arxiv.org/abs/1711.05101},
  note = {OP048: ICLR 2019（arXiv初出2017）},
  urldate = {2026-10-03},
  eprint = {1711.05101},
  archivePrefix = {arXiv}
}

@misc{OP049,
  author = {Dougal Maclaurin and David Duvenaud and Ryan P. Adams},
  title = {{Gradient-based Hyperparameter Optimization through Reversible Learning}},
  year = {2015},
  url = {https://arxiv.org/abs/1502.03492},
  note = {OP049: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {1502.03492},
  archivePrefix = {arXiv}
}

@misc{OP050,
  author = {Luca Franceschi and Michele Donini and Paolo Frasconi and Massimiliano Pontil},
  title = {{Forward and Reverse Gradient-Based Hyperparameter Optimization}},
  year = {2017},
  url = {https://arxiv.org/abs/1703.01785},
  note = {OP050: ICML 2017},
  urldate = {2026-10-03},
  eprint = {1703.01785},
  archivePrefix = {arXiv}
}

@misc{OP051,
  author = {Luca Franceschi and Paolo Frasconi and Saverio Salzo and Riccardo Grazzi and Massimilano Pontil},
  title = {{Bilevel Programming for Hyperparameter Optimization and Meta-Learning}},
  year = {2018},
  url = {https://arxiv.org/abs/1806.04910},
  note = {OP051: ICML 2018},
  urldate = {2026-10-03},
  eprint = {1806.04910},
  archivePrefix = {arXiv}
}

@misc{OP052,
  author = {Constantinos Daskalakis and Andrew Ilyas and Vasilis Syrgkanis and Haoyang Zeng},
  title = {{Training GANs with Optimism}},
  year = {2017},
  url = {https://arxiv.org/abs/1711.00141},
  note = {OP052: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {1711.00141},
  archivePrefix = {arXiv}
}

@misc{OP053,
  author = {Atilim Gunes Baydin and Robert Cornish and David Martinez Rubio and Mark Schmidt and Frank Wood},
  title = {{Online Learning Rate Adaptation with Hypergradient Descent}},
  year = {2017},
  url = {https://arxiv.org/abs/1703.04782},
  note = {OP053: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {1703.04782},
  archivePrefix = {arXiv}
}

@misc{OP054,
  author = {Max Jaderberg and Valentin Dalibard and Simon Osindero and others},
  title = {{Population Based Training of Neural Networks}},
  year = {2017},
  url = {https://arxiv.org/abs/1711.09846},
  note = {OP054: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {1711.09846},
  archivePrefix = {arXiv}
}

@misc{OP055,
  author = {Marcin Andrychowicz and Misha Denil and Sergio Gomez and others},
  title = {{Learning to learn by gradient descent by gradient descent}},
  year = {2016},
  url = {https://arxiv.org/abs/1606.04474},
  note = {OP055: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {1606.04474},
  archivePrefix = {arXiv}
}

@misc{OP056,
  author = {Irwan Bello and Barret Zoph and Vijay Vasudevan and Quoc V. Le},
  title = {{Neural Optimizer Search with Reinforcement Learning}},
  year = {2017},
  url = {https://arxiv.org/abs/1709.07417},
  note = {OP056: ICML 2017},
  urldate = {2026-10-03},
  eprint = {1709.07417},
  archivePrefix = {arXiv}
}

@misc{OP057,
  author = {Hao-Jun Michael Shi and Tsung-Hsien Lee and Shintaro Iwasaki and others},
  title = {{A Distributed Data-Parallel PyTorch Implementation of the Distributed Shampoo Optimizer for Training Neural Networks At-Scale}},
  year = {2023},
  url = {https://arxiv.org/abs/2309.06497},
  note = {OP057: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2309.06497},
  archivePrefix = {arXiv}
}

@misc{OP058,
  author = {Ziyue Liu and Ruijie Zhang and Zhengyang Wang and others},
  title = {{Muon$^2$: Boosting Muon via Adaptive Second-Moment Preconditioning}},
  year = {2026},
  url = {https://arxiv.org/abs/2604.09967},
  note = {OP058: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2604.09967},
  archivePrefix = {arXiv}
}

@misc{OP059,
  author = {Noah Amsel and Jack Zhang and Kwangjun Ahn and others},
  title = {{Dion3: Full-Stack Orthogonal Updates}},
  year = {2026},
  url = {https://arxiv.org/abs/2608.11612},
  note = {OP059: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2608.11612},
  archivePrefix = {arXiv}
}

@misc{OP060,
  author = {Dara Bahri and Hossein Mobahi and Yi Tay},
  title = {{Sharpness-Aware Minimization Improves Language Model Generalization}},
  year = {2021},
  url = {https://arxiv.org/abs/2110.08529},
  note = {OP060: ACL 2022（arXiv初出2021）},
  urldate = {2026-10-03},
  eprint = {2110.08529},
  archivePrefix = {arXiv}
}

@misc{OP061,
  author = {Haocheng Luo and Zehang Deng and Thanh-Toan Do and others},
  title = {{Sharpness-Aware Minimization in Logit Space Efficiently Enhances Direct Preference Optimization}},
  year = {2026},
  url = {https://arxiv.org/abs/2603.18258},
  note = {OP061: ICLR 2026（arXiv採択注記を確認）},
  urldate = {2026-10-03},
  eprint = {2603.18258},
  archivePrefix = {arXiv}
}

@misc{OP062,
  author = {Maximilian Mueller and Tiffany Vlaar and David Rolnick and Matthias Hein},
  title = {{Normalization Layers Are All That Sharpness-Aware Minimization Needs}},
  year = {2023},
  url = {https://arxiv.org/abs/2306.04226},
  note = {OP062: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2306.04226},
  archivePrefix = {arXiv}
}

@misc{OP063,
  author = {Kaiyi Ji and Junjie Yang and Yingbin Liang},
  title = {{Bilevel Optimization: Convergence Analysis and Enhanced Design}},
  year = {2020},
  url = {https://arxiv.org/abs/2010.07962},
  note = {OP063: ICML 2021（arXiv初出2020）},
  urldate = {2026-10-03},
  eprint = {2010.07962},
  archivePrefix = {arXiv}
}

@misc{OP064,
  author = {Riccardo Grazzi and Luca Franceschi and Massimiliano Pontil and Saverio Salzo},
  title = {{On the Iteration Complexity of Hypergradient Computation}},
  year = {2020},
  url = {https://arxiv.org/abs/2006.16218},
  note = {OP064: ICML 2020},
  urldate = {2026-10-03},
  eprint = {2006.16218},
  archivePrefix = {arXiv}
}

@misc{OP065,
  author = {Jeongyeol Kwon and Dohyun Kwon and Stephen Wright and Robert Nowak},
  title = {{A Fully First-Order Method for Stochastic Bilevel Optimization}},
  year = {2023},
  url = {https://arxiv.org/abs/2301.10945},
  note = {OP065: ICML 2023},
  urldate = {2026-10-03},
  eprint = {2301.10945},
  archivePrefix = {arXiv}
}

@misc{OP066,
  author = {Wu Lin and Scott C. Lowe and Felix Dangel and others},
  title = {{Understanding and Improving Shampoo and SOAP via Kullback-Leibler Minimization}},
  year = {2025},
  url = {https://arxiv.org/abs/2509.03378},
  note = {OP066: ICLR 2026、arXiv v10は2026-06-22拡張版},
  urldate = {2026-10-03},
  eprint = {2509.03378},
  archivePrefix = {arXiv}
}

@misc{OP067,
  author = {Stefan Horoi and Benjamin Thérien and Guy Wolf and Eugene Belilovsky},
  title = {{Can Model Merging Improve Aggregation in DiLoCo?}},
  year = {2026},
  url = {https://arxiv.org/abs/2607.03011},
  note = {OP067: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2607.03011},
  archivePrefix = {arXiv}
}

@misc{OP068,
  author = {Andrej Jovanović and Alex Iacob and Mher Safaryan and others},
  title = {{LoRDO: Distributed Low-Rank Optimization with Infrequent Communication}},
  year = {2026},
  url = {https://arxiv.org/abs/2602.04396},
  note = {OP068: ICML 2026（arXiv採択注記を確認）},
  urldate = {2026-10-03},
  eprint = {2602.04396},
  archivePrefix = {arXiv}
}

@misc{OP069,
  author = {Léon Bottou and Frank E. Curtis and Jorge Nocedal},
  title = {{Optimization Methods for Large-Scale Machine Learning}},
  year = {2016},
  url = {https://arxiv.org/abs/1606.04838},
  note = {OP069: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {1606.04838},
  archivePrefix = {arXiv}
}

@misc{OP070,
  author = {Mark Tuddenham and Adam Prügel-Bennett and Jonathan Hare},
  title = {{Orthogonalising gradients to speed up neural network optimisation}},
  year = {2022},
  url = {https://arxiv.org/abs/2202.07052},
  note = {OP070: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2202.07052},
  archivePrefix = {arXiv}
}

@misc{OP071,
  author = {Mao Ye and Bo Liu and Stephen Wright and Peter Stone and Qiang Liu},
  title = {{BOME! Bilevel Optimization Made Easy: A Simple First-Order Approach}},
  year = {2022},
  url = {https://arxiv.org/abs/2209.08709},
  note = {OP071: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2209.08709},
  archivePrefix = {arXiv}
}

@misc{OP072,
  author = {Martin Heusel and Hubert Ramsauer and Thomas Unterthiner and Bernhard Nessler and Sepp Hochreiter},
  title = {{GANs Trained by a Two Time-Scale Update Rule Converge to a Local Nash Equilibrium}},
  year = {2017},
  url = {https://arxiv.org/abs/1706.08500},
  note = {OP072: arXiv版参照（公刊状況を別途確認したものは注記） / FM076: NeurIPS 2017},
  urldate = {2026-10-03},
  eprint = {1706.08500},
  archivePrefix = {arXiv}
}

@misc{OP073,
  author = {Risheng Liu and Jiaxin Gao and Jin Zhang and Deyu Meng and Zhouchen Lin},
  title = {{Investigating Bi-Level Optimization for Learning and Vision from a Unified Perspective: A Survey and Beyond}},
  year = {2021},
  url = {https://arxiv.org/abs/2101.11517},
  note = {OP073: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2101.11517},
  archivePrefix = {arXiv}
}

@misc{OP074,
  author = {Laurent Dinh and Razvan Pascanu and Samy Bengio and Yoshua Bengio},
  title = {{Sharp Minima Can Generalize For Deep Nets}},
  year = {2017},
  url = {https://arxiv.org/abs/1703.04933},
  note = {OP074: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {1703.04933},
  archivePrefix = {arXiv}
}

@misc{OP075,
  author = {Jan Harold Alcantara and Masahiro Inoue and Akiko Takeda},
  title = {{Improved KKT Complexity for First-Order Bilevel Optimization under Weak Lower-Level Convexity}},
  year = {2026},
  url = {https://arxiv.org/abs/2609.39736},
  note = {OP075: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2609.39736},
  archivePrefix = {arXiv}
}

@misc{OP076,
  author = {Zihao Zheng and Irwin King and Songtao Lu},
  title = {{Bilevel Optimization over Saddle Points of Zero-Sum Markov Games}},
  year = {2026},
  url = {https://arxiv.org/abs/2605.26654},
  note = {OP076: ICML 2026（arXiv採択注記を確認）},
  urldate = {2026-10-03},
  eprint = {2605.26654},
  archivePrefix = {arXiv}
}

@misc{OP077,
  author = {Lev McKinney and Anvith Thudi and Juhan Bae and others},
  title = {{Gauss-Newton Unlearning for the LLM Era}},
  year = {2026},
  url = {https://arxiv.org/abs/2602.10568},
  note = {OP077: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2602.10568},
  archivePrefix = {arXiv}
}

@misc{OP078,
  author = {John Duchi and Elad Hazan and Yoram Singer},
  title = {{Adaptive Subgradient Methods for Online Learning and Stochastic Optimization}},
  year = {2011},
  url = {https://jmlr.org/papers/v12/duchi11a.html},
  note = {OP078: JMLR 12(61), 2011},
  urldate = {2026-10-03}
}

@misc{OP079,
  author = {Sashank J. Reddi and Satyen Kale and Sanjiv Kumar},
  title = {{On the Convergence of Adam and Beyond}},
  year = {2018},
  url = {https://research.google/pubs/on-the-convergence-of-adam-and-beyond/},
  note = {OP079: ICLR 2018（著者所属機関の論文ページ）},
  urldate = {2026-10-03}
}

@misc{OP080,
  author = {Shun-ichi Amari},
  title = {{Natural Gradient Works Efficiently in Learning}},
  year = {1998},
  url = {https://doi.org/10.1162/089976698300017746},
  note = {OP080: Neural Computation 10(2):251–276, 1998},
  urldate = {2026-10-03}
}

@misc{OP081,
  author = {Magnus R. Hestenes and Eduard Stiefel},
  title = {{Methods of Conjugate Gradients for Solving Linear Systems}},
  year = {1952},
  url = {https://nvlpubs.nist.gov/nistpubs/jres/049/jresv49n6p409_a1b.pdf},
  note = {OP081: Journal of Research of the National Bureau of Standards 49(6):409–436, 1952},
  urldate = {2026-10-03}
}

@misc{OP082,
  author = {Kenneth Levenberg},
  title = {{A Method for the Solution of Certain Non-Linear Problems in Least Squares}},
  year = {1944},
  url = {https://doi.org/10.1090/qam/10666},
  note = {OP082: Quarterly of Applied Mathematics 2(2):164–168, 1944。原典アクセス403、Marquardt出版社参考文献と書誌を照合},
  urldate = {2026-10-03}
}

@misc{OP083,
  author = {Donald W. Marquardt},
  title = {{An Algorithm for Least-Squares Estimation of Nonlinear Parameters}},
  year = {1963},
  url = {https://epubs.siam.org/doi/10.1137/0111030},
  note = {OP083: Journal of the Society for Industrial and Applied Mathematics 11(2):431–441, 1963。2006は電子公開年},
  urldate = {2026-10-03}
}

@misc{OP084,
  author = {James Martens},
  title = {{Deep Learning via Hessian-Free Optimization}},
  year = {2010},
  url = {https://icml.cc/2010/papers/458.pdf},
  note = {OP084: ICML 2010},
  urldate = {2026-10-03}
}

@misc{OP085,
  author = {David Carlson and Volkan Cevher and Lawrence Carin},
  title = {{Stochastic Spectral Descent for Restricted Boltzmann Machines}},
  year = {2015},
  url = {https://proceedings.mlr.press/v38/carlson15.html},
  note = {OP085: AISTATS 2015, PMLR 38:111–119},
  urldate = {2026-10-03}
}

@misc{OP086,
  author = {Keller Jordan},
  title = {{Muon: An optimizer for hidden layers in neural networks}},
  year = {2024},
  url = {https://kellerjordan.github.io/posts/muon/},
  note = {OP086: 提案者の技術記事、2024-12-08（2025追記あり、査読論文ではない）},
  urldate = {2026-10-03}
}

@misc{OP087,
  author = {Jungmin Kwon and Jeongseop Kim and Hyunseo Park and In Kwon Choi},
  title = {{ASAM: Adaptive Sharpness-Aware Minimization for Scale-Invariant Learning of Deep Neural Networks}},
  year = {2021},
  url = {https://proceedings.mlr.press/v139/kwon21b.html},
  note = {OP087: ICML 2021, PMLR 139:5905–5914},
  urldate = {2026-10-03}
}

@misc{OP088,
  author = {Hong Liu and Zhiyuan Li and David Hall and Percy Liang and Tengyu Ma},
  title = {{Sophia: A Scalable Stochastic Second-order Optimizer for Language Model Pre-training}},
  year = {2023},
  url = {https://arxiv.org/abs/2305.14342},
  note = {OP088: arXiv版参照},
  urldate = {2026-10-03},
  eprint = {2305.14342},
  archivePrefix = {arXiv}
}

@misc{OP089,
  author = {Konstantin Mishchenko and Aaron Defazio},
  title = {{Prodigy: An Expeditiously Adaptive Parameter-Free Learner}},
  year = {2023},
  url = {https://arxiv.org/abs/2306.06101},
  note = {OP089: arXiv版参照},
  urldate = {2026-10-03},
  eprint = {2306.06101},
  archivePrefix = {arXiv}
}

@misc{OP090,
  author = {Shohei Taniguchi and Keno Harada and Gouki Minegishi and others},
  title = {{ADOPT: Modified Adam Can Converge with Any β₂ with the Optimal Rate}},
  year = {2024},
  url = {https://arxiv.org/abs/2411.02853},
  note = {OP090: NeurIPS 2024（arXiv Commentsで確認）},
  urldate = {2026-10-03},
  eprint = {2411.02853},
  archivePrefix = {arXiv}
}

@misc{OP091,
  author = {Xi-Lin Li},
  title = {{Preconditioned Stochastic Gradient Descent}},
  year = {2015},
  url = {https://arxiv.org/abs/1512.04202},
  note = {OP091: arXiv初出2015、TNNLS版へのDOIあり},
  urldate = {2026-10-03},
  eprint = {1512.04202},
  archivePrefix = {arXiv}
}

@misc{GE01,
  author = {Masashi Sugiyama and Matthias Krauledat and Klaus-Robert Müller},
  title = {{Covariate Shift Adaptation by Importance Weighted Cross Validation}},
  year = {2007},
  url = {https://www.jmlr.org/papers/v8/sugiyama07a.html},
  note = {GE01: JMLR 2007},
  urldate = {2026-10-03}
}

@misc{GE02,
  author = {Zachary C. Lipton and Yu-Xiang Wang and Alexander J. Smola},
  title = {{Detecting and Correcting for Label Shift with Black Box Predictors}},
  year = {2018},
  url = {https://proceedings.mlr.press/v80/lipton18a.html},
  note = {GE02: ICML 2018},
  urldate = {2026-10-03}
}

@misc{GE03,
  author = {Shai Ben-David and John Blitzer and Koby Crammer and Alex Kulesza and Fernando Pereira and Jennifer Wortman Vaughan},
  title = {{A theory of learning from different domains}},
  year = {2009},
  url = {https://link.springer.com/article/10.1007/s10994-009-5152-4},
  note = {GE03: online 2009; Machine Learning 2010},
  urldate = {2026-10-03}
}

@misc{GE04,
  author = {Yaroslav Ganin and Evgeniya Ustinova and Hana Ajakan and Pascal Germain and Hugo Larochelle and François Laviolette and Mario Marchand and Victor Lempitsky},
  title = {{Domain-Adversarial Training of Neural Networks}},
  year = {2015},
  url = {https://jmlr.org/papers/v17/15-239.html},
  note = {GE04: arXiv 2015; JMLR 2016},
  urldate = {2026-10-03}
}

@misc{GE05,
  author = {Martin Arjovsky and Léon Bottou and Ishaan Gulrajani and David Lopez-Paz},
  title = {{Invariant Risk Minimization}},
  year = {2019},
  url = {https://arxiv.org/abs/1907.02893},
  note = {GE05: arXiv preprint},
  urldate = {2026-10-03},
  eprint = {1907.02893},
  archivePrefix = {arXiv}
}

@misc{GE06,
  author = {David Krueger and Ethan Caballero and Joern-Henrik Jacobsen and Amy Zhang and Jonathan Binas and Dinghuai Zhang and Remi Le Priol and Aaron Courville},
  title = {{Out-of-Distribution Generalization via Risk Extrapolation (REx)}},
  year = {2020},
  url = {https://proceedings.mlr.press/v139/krueger21a.html},
  note = {GE06: arXiv 2020; ICML 2021},
  urldate = {2026-10-03}
}

@misc{GE07,
  author = {Shiori Sagawa and Pang Wei Koh and Tatsunori B. Hashimoto and Percy Liang},
  title = {{Distributionally Robust Neural Networks for Group Shifts: On the Importance of Regularization for Worst-Case Generalization}},
  year = {2019},
  url = {https://arxiv.org/abs/1911.08731},
  note = {GE07: arXiv 2019; ICLR 2020},
  urldate = {2026-10-03},
  eprint = {1911.08731},
  archivePrefix = {arXiv}
}

@misc{GE08,
  author = {Ishaan Gulrajani and David Lopez-Paz},
  title = {{In Search of Lost Domain Generalization}},
  year = {2020},
  url = {https://arxiv.org/abs/2007.01434},
  note = {GE08: arXiv 2020; ICLR 2021},
  urldate = {2026-10-03},
  eprint = {2007.01434},
  archivePrefix = {arXiv}
}

@misc{GE09,
  author = {Pang Wei Koh and Shiori Sagawa and others},
  title = {{WILDS: A Benchmark of in-the-Wild Distribution Shifts}},
  year = {2020},
  url = {https://arxiv.org/abs/2012.07421},
  note = {GE09: arXiv 2020; ICML 2021},
  urldate = {2026-10-03},
  eprint = {2012.07421},
  archivePrefix = {arXiv}
}

@misc{GE10,
  author = {Dan Hendrycks and Thomas Dietterich},
  title = {{Benchmarking Neural Network Robustness to Common Corruptions and Perturbations}},
  year = {2019},
  url = {https://arxiv.org/abs/1903.12261},
  note = {GE10: ICLR 2019},
  urldate = {2026-10-03},
  eprint = {1903.12261},
  archivePrefix = {arXiv}
}

@misc{GE11,
  author = {Alexandre Rame and Corentin Dancette and Matthieu Cord},
  title = {{Fishr: Invariant Gradient Variances for Out-of-Distribution Generalization}},
  year = {2021},
  url = {https://proceedings.mlr.press/v162/rame22a.html},
  note = {GE11: arXiv 2021; ICML 2022},
  urldate = {2026-10-03}
}

@misc{GE12,
  author = {Mitchell Wortsman and Gabriel Ilharco and others},
  title = {{Robust fine-tuning of zero-shot models}},
  year = {2021},
  url = {https://arxiv.org/abs/2109.01903},
  note = {GE12: arXiv 2021; CVPR 2022},
  urldate = {2026-10-03},
  eprint = {2109.01903},
  archivePrefix = {arXiv}
}

@misc{GE13,
  author = {Mitchell Wortsman and Gabriel Ilharco and others},
  title = {{Model soups: averaging weights of multiple fine-tuned models improves accuracy without increasing inference time}},
  year = {2022},
  url = {https://arxiv.org/abs/2203.05482},
  note = {GE13: ICML 2022 / SY012: ICML 2022},
  urldate = {2026-10-03},
  eprint = {2203.05482},
  archivePrefix = {arXiv}
}

@misc{GE14,
  author = {Dequan Wang and Evan Shelhamer and Shaoteng Liu and Bruno Olshausen and Trevor Darrell},
  title = {{Tent: Fully Test-time Adaptation by Entropy Minimization}},
  year = {2020},
  url = {https://arxiv.org/abs/2006.10726},
  note = {GE14: arXiv 2020; ICLR 2021},
  urldate = {2026-10-03},
  eprint = {2006.10726},
  archivePrefix = {arXiv}
}

@misc{GE15,
  author = {Qin Wang and Olga Fink and Luc Van Gool and Dengxin Dai},
  title = {{Continual Test-Time Domain Adaptation}},
  year = {2022},
  url = {https://arxiv.org/abs/2203.13591},
  note = {GE15: CVPR 2022},
  urldate = {2026-10-03},
  eprint = {2203.13591},
  archivePrefix = {arXiv}
}

@misc{GE16,
  author = {Shuaicheng Niu and Jiaxiang Wu and Yifan Zhang and Yaofo Chen and Shijian Zheng and Peilin Zhao and Mingkui Tan},
  title = {{Efficient Test-Time Model Adaptation without Forgetting}},
  year = {2022},
  url = {https://proceedings.mlr.press/v162/niu22a.html},
  note = {GE16: ICML 2022},
  urldate = {2026-10-03}
}

@misc{GE17,
  author = {Shuaicheng Niu and Jiaxiang Wu and Yifan Zhang and Zhiquan Wen and Yaofo Chen and Peilin Zhao and Mingkui Tan},
  title = {{Towards Stable Test-Time Adaptation in Dynamic Wild World}},
  year = {2023},
  url = {https://arxiv.org/abs/2302.12400},
  note = {GE17: ICLR 2023},
  urldate = {2026-10-03},
  eprint = {2302.12400},
  archivePrefix = {arXiv}
}

@misc{GE18,
  author = {Hao Zhao and Yuejiang Liu and Alexandre Alahi and Tao Lin},
  title = {{On Pitfalls of Test-Time Adaptation}},
  year = {2023},
  url = {https://arxiv.org/abs/2306.03536},
  note = {GE18: ICML 2023},
  urldate = {2026-10-03},
  eprint = {2306.03536},
  archivePrefix = {arXiv}
}

@misc{GE19,
  author = {Zehao Xiao and Cees G. M. Snoek},
  title = {{Beyond Model Adaptation at Test Time: A Survey}},
  year = {2024},
  url = {https://arxiv.org/abs/2411.03687},
  note = {GE19: arXiv preprint},
  urldate = {2026-10-03},
  eprint = {2411.03687},
  archivePrefix = {arXiv}
}

@misc{GE20,
  author = {Dan Hendrycks and Kevin Gimpel},
  title = {{A Baseline for Detecting Misclassified and Out-of-Distribution Examples in Neural Networks}},
  year = {2016},
  url = {https://arxiv.org/abs/1610.02136},
  note = {GE20: arXiv 2016; ICLR 2017},
  urldate = {2026-10-03},
  eprint = {1610.02136},
  archivePrefix = {arXiv}
}

@misc{GE21,
  author = {Shiyu Liang and Yixuan Li and R. Srikant},
  title = {{Enhancing The Reliability of Out-of-distribution Image Detection in Neural Networks}},
  year = {2017},
  url = {https://arxiv.org/abs/1706.02690},
  note = {GE21: arXiv 2017; ICLR 2018},
  urldate = {2026-10-03},
  eprint = {1706.02690},
  archivePrefix = {arXiv}
}

@misc{GE22,
  author = {Dan Hendrycks and Mantas Mazeika and Thomas Dietterich},
  title = {{Deep Anomaly Detection with Outlier Exposure}},
  year = {2018},
  url = {https://arxiv.org/abs/1812.04606},
  note = {GE22: arXiv 2018; ICLR 2019},
  urldate = {2026-10-03},
  eprint = {1812.04606},
  archivePrefix = {arXiv}
}

@misc{GE23,
  author = {Weitang Liu and Xiaoyun Wang and John D. Owens and Yixuan Li},
  title = {{Energy-based Out-of-distribution Detection}},
  year = {2020},
  url = {https://arxiv.org/abs/2010.03759},
  note = {GE23: NeurIPS 2020},
  urldate = {2026-10-03},
  eprint = {2010.03759},
  archivePrefix = {arXiv}
}

@misc{GE24,
  author = {Hongxin Wei and Renchunzi Xie and Hao Cheng and Lei Feng and Bo An and Yixuan Li},
  title = {{Mitigating Neural Network Overconfidence with Logit Normalization}},
  year = {2022},
  url = {https://arxiv.org/abs/2205.09310},
  note = {GE24: ICML 2022},
  urldate = {2026-10-03},
  eprint = {2205.09310},
  archivePrefix = {arXiv}
}

@misc{GE25,
  author = {Jingkang Yang and Pengyun Wang and others},
  title = {{OpenOOD: Benchmarking Generalized Out-of-Distribution Detection}},
  year = {2022},
  url = {https://arxiv.org/abs/2210.07242},
  note = {GE25: NeurIPS 2022 Datasets and Benchmarks},
  urldate = {2026-10-03},
  eprint = {2210.07242},
  archivePrefix = {arXiv}
}

@misc{GE26,
  author = {Jingyang Zhang and Jingkang Yang and others},
  title = {{OpenOOD v1.5: Enhanced Benchmark for Out-of-Distribution Detection}},
  year = {2023},
  url = {https://arxiv.org/abs/2306.09301},
  note = {GE26: DMLR採択表示（arXiv v5, 2024-12-16）},
  urldate = {2026-10-03},
  eprint = {2306.09301},
  archivePrefix = {arXiv}
}

@misc{GE27,
  author = {Jingkang Yang and Kaiyang Zhou and Yixuan Li and Ziwei Liu},
  title = {{Generalized Out-of-Distribution Detection: A Survey}},
  year = {2021},
  url = {https://arxiv.org/abs/2110.11334},
  note = {GE27: arXiv survey（2024改訂版確認）},
  urldate = {2026-10-03},
  eprint = {2110.11334},
  archivePrefix = {arXiv}
}

@misc{GE28,
  author = {Chuan Guo and Geoff Pleiss and Yu Sun and Kilian Q. Weinberger},
  title = {{On Calibration of Modern Neural Networks}},
  year = {2017},
  url = {https://proceedings.mlr.press/v70/guo17a.html},
  note = {GE28: ICML 2017},
  urldate = {2026-10-03}
}

@misc{GE29,
  author = {Tilmann Gneiting and Adrian E. Raftery},
  title = {{Strictly Proper Scoring Rules, Prediction, and Estimation}},
  year = {2007},
  url = {https://doi.org/10.1198/016214506000001437},
  note = {GE29: JASA 2007},
  urldate = {2026-10-03}
}

@misc{GE30,
  author = {Jeremy Nixon and Mike Dusenberry and Ghassen Jerfel and Timothy Nguyen and Jeremiah Liu and Linchuan Zhang and Dustin Tran},
  title = {{Measuring Calibration in Deep Learning}},
  year = {2019},
  url = {https://arxiv.org/abs/1904.01685},
  note = {GE30: arXiv preprint（2020改訂）},
  urldate = {2026-10-03},
  eprint = {1904.01685},
  archivePrefix = {arXiv}
}

@misc{GE31,
  author = {Aviral Kumar and Sunita Sarawagi and Ujjwal Jain},
  title = {{Trainable Calibration Measures for Neural Networks from Kernel Mean Embeddings}},
  year = {2018},
  url = {https://proceedings.mlr.press/v80/kumar18a.html},
  note = {GE31: ICML 2018},
  urldate = {2026-10-03}
}

@misc{GE32,
  author = {Ananya Kumar and Percy Liang and Tengyu Ma},
  title = {{Verified Uncertainty Calibration}},
  year = {2019},
  url = {https://arxiv.org/abs/1909.10155},
  note = {GE32: NeurIPS 2019},
  urldate = {2026-10-03},
  eprint = {1909.10155},
  archivePrefix = {arXiv}
}

@misc{GE33,
  author = {Jarosław Błasiok and Parikshit Gopalan and Lunjia Hu and Preetum Nakkiran},
  title = {{A Unifying Theory of Distance from Calibration}},
  year = {2022},
  url = {https://arxiv.org/abs/2211.16886},
  note = {GE33: arXiv 2022; STOC 2023},
  urldate = {2026-10-03},
  eprint = {2211.16886},
  archivePrefix = {arXiv}
}

@misc{GE34,
  author = {Meelis Kull and Miquel Perello-Nieto and Markus Kängsepp and Telmo Silva Filho and Hao Song and Peter Flach},
  title = {{Beyond temperature scaling: Obtaining well-calibrated multiclass probabilities with Dirichlet calibration}},
  year = {2019},
  url = {https://arxiv.org/abs/1910.12656},
  note = {GE34: NeurIPS 2019},
  urldate = {2026-10-03},
  eprint = {1910.12656},
  archivePrefix = {arXiv}
}

@misc{GE35,
  author = {Jishnu Mukhoti and Viveka Kulharia and Amartya Sanyal and Stuart Golodetz and Philip H. S. Torr and Puneet K. Dokania},
  title = {{Calibrating Deep Neural Networks using Focal Loss}},
  year = {2020},
  url = {https://arxiv.org/abs/2002.09437},
  note = {GE35: NeurIPS 2020},
  urldate = {2026-10-03},
  eprint = {2002.09437},
  archivePrefix = {arXiv}
}

@misc{GE36,
  author = {Matthias Minderer and Josip Djolonga and Rob Romijnders and Frances Hubis and Xiaohua Zhai and Neil Houlsby and Dustin Tran and Mario Lucic},
  title = {{Revisiting the Calibration of Modern Neural Networks}},
  year = {2021},
  url = {https://arxiv.org/abs/2106.07998},
  note = {GE36: NeurIPS 2021},
  urldate = {2026-10-03},
  eprint = {2106.07998},
  archivePrefix = {arXiv}
}

@misc{GE37,
  author = {Balaji Lakshminarayanan and Alexander Pritzel and Charles Blundell},
  title = {{Simple and Scalable Predictive Uncertainty Estimation using Deep Ensembles}},
  year = {2016},
  url = {https://arxiv.org/abs/1612.01474},
  note = {GE37: arXiv 2016; NeurIPS 2017},
  urldate = {2026-10-03},
  eprint = {1612.01474},
  archivePrefix = {arXiv}
}

@misc{GE38,
  author = {Yarin Gal and Zoubin Ghahramani},
  title = {{Dropout as a Bayesian Approximation: Representing Model Uncertainty in Deep Learning}},
  year = {2015},
  url = {https://arxiv.org/abs/1506.02142},
  note = {GE38: arXiv 2015; ICML 2016},
  urldate = {2026-10-03},
  eprint = {1506.02142},
  archivePrefix = {arXiv}
}

@misc{GE39,
  author = {Yaniv Ovadia and Emily Fertig and Jie Ren and Zachary Nado and D Sculley and Sebastian Nowozin and Joshua V. Dillon and Balaji Lakshminarayanan and Jasper Snoek},
  title = {{Can You Trust Your Model's Uncertainty? Evaluating Predictive Uncertainty Under Dataset Shift}},
  year = {2019},
  url = {https://arxiv.org/abs/1906.02530},
  note = {GE39: NeurIPS 2019},
  urldate = {2026-10-03},
  eprint = {1906.02530},
  archivePrefix = {arXiv}
}

@misc{GE40,
  author = {Yoav Wald and Amir Feder and Daniel Greenfeld and Uri Shalit},
  title = {{On Calibration and Out-of-domain Generalization}},
  year = {2021},
  url = {https://arxiv.org/abs/2102.10395},
  note = {GE40: NeurIPS 2021},
  urldate = {2026-10-03},
  eprint = {2102.10395},
  archivePrefix = {arXiv}
}

@misc{GE41,
  author = {Volodymyr Kuleshov and Nathan Fenner and Stefano Ermon},
  title = {{Accurate Uncertainties for Deep Learning Using Calibrated Regression}},
  year = {2018},
  url = {https://arxiv.org/abs/1807.00263},
  note = {GE41: ICML 2018},
  urldate = {2026-10-03},
  eprint = {1807.00263},
  archivePrefix = {arXiv}
}

@misc{GE42,
  author = {Anastasios N. Angelopoulos and Stephen Bates},
  title = {{A Gentle Introduction to Conformal Prediction and Distribution-Free Uncertainty Quantification}},
  year = {2021},
  url = {https://arxiv.org/abs/2107.07511},
  note = {GE42: arXiv tutorial},
  urldate = {2026-10-03},
  eprint = {2107.07511},
  archivePrefix = {arXiv}
}

@misc{GE43,
  author = {Ryan J. Tibshirani and Rina Foygel Barber and Emmanuel J. Candès and Aaditya Ramdas},
  title = {{Conformal Prediction Under Covariate Shift}},
  year = {2019},
  url = {https://arxiv.org/abs/1904.06019},
  note = {GE43: NeurIPS 2019},
  urldate = {2026-10-03},
  eprint = {1904.06019},
  archivePrefix = {arXiv}
}

@misc{GE44,
  author = {Isaac Gibbs and Emmanuel Candès},
  title = {{Adaptive Conformal Inference Under Distribution Shift}},
  year = {2021},
  url = {https://arxiv.org/abs/2106.00170},
  note = {GE44: NeurIPS 2021},
  urldate = {2026-10-03},
  eprint = {2106.00170},
  archivePrefix = {arXiv}
}

@misc{GE45,
  author = {Anastasios N. Angelopoulos and Stephen Bates and Adam Fisch and Lihua Lei and Tal Schuster},
  title = {{Conformal Risk Control}},
  year = {2022},
  url = {https://arxiv.org/abs/2208.02814},
  note = {GE45: arXiv 2022; ICLR 2024},
  urldate = {2026-10-03},
  eprint = {2208.02814},
  archivePrefix = {arXiv}
}

@misc{GE46,
  author = {Rina Foygel Barber and Emmanuel J. Candès and Aaditya Ramdas and Ryan J. Tibshirani},
  title = {{The limits of distribution-free conditional predictive inference}},
  year = {2019},
  url = {https://arxiv.org/abs/1903.04684},
  note = {GE46: arXiv 2019; Information and Inference 2021},
  urldate = {2026-10-03},
  eprint = {1903.04684},
  archivePrefix = {arXiv}
}

@misc{GE47,
  author = {Lars Van Der Laan and Ahmed Alaa},
  title = {{Generalized Venn and Venn-Abers Calibration with Applications in Conformal Prediction}},
  year = {2025},
  url = {https://proceedings.mlr.press/v267/van-der-laan25a.html},
  note = {GE47: ICML 2025},
  urldate = {2026-10-03}
}

@misc{GE48,
  author = {Moritz Hardt and Eric Price and Nathan Srebro},
  title = {{Equality of Opportunity in Supervised Learning}},
  year = {2016},
  url = {https://arxiv.org/abs/1610.02413},
  note = {GE48: NeurIPS 2016},
  urldate = {2026-10-03},
  eprint = {1610.02413},
  archivePrefix = {arXiv}
}

@misc{GE49,
  author = {Jon Kleinberg and Sendhil Mullainathan and Manish Raghavan},
  title = {{Inherent Trade-Offs in the Fair Determination of Risk Scores}},
  year = {2016},
  url = {https://arxiv.org/abs/1609.05807},
  note = {GE49: arXiv 2016; ITCS 2017},
  urldate = {2026-10-03},
  eprint = {1609.05807},
  archivePrefix = {arXiv}
}

@misc{GE50,
  author = {Geoff Pleiss and Manish Raghavan and Felix Wu and Jon Kleinberg and Kilian Q. Weinberger},
  title = {{On Fairness and Calibration}},
  year = {2017},
  url = {https://arxiv.org/abs/1709.02012},
  note = {GE50: NeurIPS 2017},
  urldate = {2026-10-03},
  eprint = {1709.02012},
  archivePrefix = {arXiv}
}

@misc{GE51,
  author = {Úrsula Hébert-Johnson and Michael P. Kim and Omer Reingold and Guy N. Rothblum},
  title = {{Multicalibration: Calibration for the (Computationally-Identifiable) Masses}},
  year = {2017},
  url = {https://proceedings.mlr.press/v80/hebert-johnson18a.html},
  note = {GE51: arXiv 2017（旧題）; ICML 2018},
  urldate = {2026-10-03}
}

@misc{GE52,
  author = {Alekh Agarwal and Alina Beygelzimer and Miroslav Dudík and John Langford and Hanna Wallach},
  title = {{A Reductions Approach to Fair Classification}},
  year = {2018},
  url = {https://proceedings.mlr.press/v80/agarwal18a.html},
  note = {GE52: ICML 2018},
  urldate = {2026-10-03}
}

@misc{GE53,
  author = {Joy Buolamwini and Timnit Gebru},
  title = {{Gender Shades: Intersectional Accuracy Disparities in Commercial Gender Classification}},
  year = {2018},
  url = {https://proceedings.mlr.press/v81/buolamwini18a.html},
  note = {GE53: FAT* 2018},
  urldate = {2026-10-03}
}

@misc{GE54,
  author = {Solon Barocas and Moritz Hardt and Arvind Narayanan},
  title = {{Fairness and Machine Learning: Limitations and Opportunities}},
  year = {2023},
  url = {https://fairmlbook.org/},
  note = {GE54: MIT Press 2023（書籍版）},
  urldate = {2026-10-03}
}

@misc{GE55,
  author = {Beepul Bharti and Mary Versa Clemens-Sewall and Paul Yi and Jeremias Sulam},
  title = {{Multiaccuracy and Multicalibration via Proxy Groups}},
  year = {2025},
  url = {https://proceedings.mlr.press/v267/bharti25a.html},
  note = {GE55: ICML 2025},
  urldate = {2026-10-03}
}

@misc{GE56,
  author = {Lunjia Hu and Haipeng Luo and Spandan Senapati and Vatsal Sharan},
  title = {{Efficient Swap Multicalibration of Elicitable Properties}},
  year = {2026},
  url = {https://proceedings.mlr.press/v336/hu26b.html},
  note = {GE56: COLT 2026},
  urldate = {2026-10-03}
}

@misc{GE57,
  author = {Brenden M. Lake and Marco Baroni},
  title = {{Generalization without systematicity: On the compositional skills of sequence-to-sequence recurrent networks}},
  year = {2017},
  url = {https://arxiv.org/abs/1711.00350},
  note = {GE57: arXiv 2017; ICML 2018},
  urldate = {2026-10-03},
  eprint = {1711.00350},
  archivePrefix = {arXiv}
}

@misc{GE58,
  author = {Daniel Keysers and Nathanael Schärli and others},
  title = {{Measuring Compositional Generalization: A Comprehensive Method on Realistic Data}},
  year = {2019},
  url = {https://arxiv.org/abs/1912.09713},
  note = {GE58: arXiv 2019; ICLR 2020},
  urldate = {2026-10-03},
  eprint = {1912.09713},
  archivePrefix = {arXiv}
}

@misc{GE59,
  author = {Najoung Kim and Tal Linzen},
  title = {{COGS: A Compositional Generalization Challenge Based on Semantic Interpretation}},
  year = {2020},
  url = {https://aclanthology.org/2020.emnlp-main.731/},
  note = {GE59: EMNLP 2020},
  urldate = {2026-10-03}
}

@misc{GE60,
  author = {Zhengxuan Wu and Christopher D. Manning and Christopher Potts},
  title = {{ReCOGS: How Incidental Details of a Logical Form Overshadow an Evaluation of Semantic Interpretation}},
  year = {2023},
  url = {https://aclanthology.org/2023.tacl-1.96/},
  note = {GE60: TACL 2023},
  urldate = {2026-10-03}
}

@misc{GE61,
  author = {Brenden M. Lake and Marco Baroni},
  title = {{Human-like systematic generalization through a meta-learning neural network}},
  year = {2023},
  url = {https://www.nature.com/articles/s41586-023-06668-3},
  note = {GE61: Nature 2023},
  urldate = {2026-10-03}
}

@misc{GE62,
  author = {Ahmad Jabbar and Cleo Condoravdi and Christopher Potts},
  title = {{Distinguishing fair from unfair compositional generalization tasks}},
  year = {2025},
  url = {https://aclanthology.org/2025.findings-emnlp.1133/},
  note = {GE62: Findings of EMNLP 2025},
  urldate = {2026-10-03}
}

@misc{GE63,
  author = {Ziyao Xu and Cong Wang and Houfeng Wang},
  title = {{Investigating More Explainable and Partition-Free Compositionality Estimation for LLMs: A Rule-Generation Perspective}},
  year = {2026},
  url = {https://aclanthology.org/2026.acl-long.409/},
  note = {GE63: ACL 2026},
  urldate = {2026-10-03}
}

@misc{GE64,
  author = {Arnas Uselis and Andrea Dittadi and Seong Joon Oh},
  title = {{Compositional Generalization Requires Linear, Orthogonal Representations in Vision Embedding Models}},
  year = {2026},
  url = {https://arxiv.org/abs/2602.24264},
  note = {GE64: ICML 2026（arXiv表示）},
  urldate = {2026-10-03},
  eprint = {2602.24264},
  archivePrefix = {arXiv}
}

@misc{GE65,
  author = {Nived Rajaraman and Audrey Huang and Miroslav Dudik and Robert Schapire and Dylan Foster and Akshay Krishnamurthy},
  title = {{Learning to Reason with Curriculum II: Compositional Generalization}},
  year = {2026},
  url = {https://arxiv.org/abs/2606.27721},
  note = {GE65: arXiv preprint, 2026-06-26},
  urldate = {2026-10-03},
  eprint = {2606.27721},
  archivePrefix = {arXiv}
}

@misc{GE66,
  author = {Hirotugu Akaike},
  title = {{A new look at the statistical model identification}},
  year = {1974},
  url = {https://doi.org/10.1109/TAC.1974.1100705},
  note = {GE66: IEEE Transactions on Automatic Control 1974},
  urldate = {2026-10-03}
}

@misc{GE67,
  author = {竹内啓},
  title = {{情報統計量の分布とモデルの適切さの規準}},
  year = {1976},
  url = {https://ndlsearch.ndl.go.jp/books/R000000004-I1701125},
  note = {GE67: 数理科学 14(3), 12–18, 1976},
  urldate = {2026-10-03}
}

@misc{GE68,
  author = {Sumio Watanabe},
  title = {{Asymptotic Equivalence of Bayes Cross Validation and Widely Applicable Information Criterion in Singular Learning Theory}},
  year = {2010},
  url = {https://www.jmlr.org/papers/v11/watanabe10a.html},
  note = {GE68: JMLR 2010},
  urldate = {2026-10-03}
}

@misc{GE69,
  author = {Aki Vehtari and Andrew Gelman and Jonah Gabry},
  title = {{Practical Bayesian model evaluation using leave-one-out cross-validation and WAIC}},
  year = {2015},
  url = {https://arxiv.org/abs/1507.04544},
  note = {GE69: arXiv 2015; Statistics and Computing 2017},
  urldate = {2026-10-03},
  eprint = {1507.04544},
  archivePrefix = {arXiv}
}

@misc{GE70,
  author = {Valentin Thomas and Fabian Pedregosa and Bart van Merriënboer and Pierre-Antoine Mangazol and Yoshua Bengio and Nicolas Le Roux},
  title = {{On the interplay between noise and curvature and its effect on optimization and generalization}},
  year = {2019},
  url = {https://arxiv.org/abs/1906.07774},
  note = {GE70: arXiv 2019; AISTATS 2020 / MD08: AISTATS 2020, published (preprint 2019)},
  urldate = {2026-10-03},
  eprint = {1906.07774},
  archivePrefix = {arXiv}
}

@misc{GE71,
  author = {Saurav Kadavath and Tom Conerly and others},
  title = {{Language Models (Mostly) Know What They Know}},
  year = {2022},
  url = {https://arxiv.org/abs/2207.05221},
  note = {GE71: arXiv preprint},
  urldate = {2026-10-03},
  eprint = {2207.05221},
  archivePrefix = {arXiv}
}

@misc{GE72,
  author = {Lorenz Kuhn and Yarin Gal and Sebastian Farquhar},
  title = {{Semantic Uncertainty: Linguistic Invariances for Uncertainty Estimation in Natural Language Generation}},
  year = {2023},
  url = {https://arxiv.org/abs/2302.09664},
  note = {GE72: ICLR 2023},
  urldate = {2026-10-03},
  eprint = {2302.09664},
  archivePrefix = {arXiv}
}

@misc{GE73,
  author = {Chiwei Zhu and Benfeng Xu and Quan Wang and Yongdong Zhang and Zhendong Mao},
  title = {{On the Calibration of Large Language Models and Alignment}},
  year = {2023},
  url = {https://arxiv.org/abs/2311.13240},
  note = {GE73: Findings of EMNLP 2023（arXiv表示）},
  urldate = {2026-10-03},
  eprint = {2311.13240},
  archivePrefix = {arXiv}
}

@misc{GE74,
  author = {Qing Lyu and Kumar Shridhar and Chaitanya Malaviya and Li Zhang and Yanai Elazar and Niket Tandon and Marianna Apidianaki and Mrinmaya Sachan and Chris Callison-Burch},
  title = {{Calibrating Large Language Models with Sample Consistency}},
  year = {2024},
  url = {https://arxiv.org/abs/2402.13904},
  note = {GE74: AAAI 2024（arXiv v2表示、2026-02-23改訂）},
  urldate = {2026-10-03},
  eprint = {2402.13904},
  archivePrefix = {arXiv}
}

@misc{GE75,
  author = {Mohammad Anas Jawad and Cornelia Caragea},
  title = {{CaliDist: Calibrating Large Language Models via Behavioral Robustness to Distraction}},
  year = {2026},
  url = {https://proceedings.mlr.press/v306/jawad26a.html},
  note = {GE75: ICML 2026},
  urldate = {2026-10-03}
}

@misc{GE76,
  author = {Eunyi Lyou and Yunjeong Choi and Junho Lee and Joonseok Lee},
  title = {{Domain Generalization via Text-Anchored Information Bottleneck}},
  year = {2026},
  url = {https://arxiv.org/abs/2607.01657},
  note = {GE76: ECCV 2026採択（arXiv表示）, 2026-07-02},
  urldate = {2026-10-03},
  eprint = {2607.01657},
  archivePrefix = {arXiv}
}

@misc{GE77,
  author = {Tien-Hung Nguyen and Tien-Dat Tran and M.-Duong Nguyen and Kok-Seng Wong},
  title = {{Learning Subset-Shared Invariances for Domain Generalization with Mixture-of-Experts}},
  year = {2026},
  url = {https://arxiv.org/abs/2606.25665},
  note = {GE77: arXiv preprint, 2026-06-24},
  urldate = {2026-10-03},
  eprint = {2606.25665},
  archivePrefix = {arXiv}
}

@misc{GE78,
  author = {Sravan Danda and Aditya Challa and Shlok Mehendale and Snehanshu Saha},
  title = {{Matching High-Dimensional Geometric Quantiles for Test-Time Adaptation of Transformers and Convolutional Networks Alike}},
  year = {2026},
  url = {https://arxiv.org/abs/2601.11022},
  note = {GE78: arXiv preprint, 2026-01-16},
  urldate = {2026-10-03},
  eprint = {2601.11022},
  archivePrefix = {arXiv}
}

@misc{GE79,
  author = {Gerhard Krumpl and Henning Avenhaus and Horst Possegger},
  title = {{One Model, Many Behaviors: Training-Induced Effects on Out-of-Distribution Detection}},
  year = {2026},
  url = {https://arxiv.org/abs/2601.10836},
  note = {GE79: WACV 2026, arXiv 2026-01-15},
  urldate = {2026-10-03},
  eprint = {2601.10836},
  archivePrefix = {arXiv}
}

@misc{GE80,
  author = {Yingkai Yang and Chaoqi Chen and Hui Huang},
  title = {{Back to Source: Open-Set Continual Test-Time Adaptation via Domain Compensation}},
  year = {2026},
  url = {https://arxiv.org/abs/2604.21772},
  note = {GE80: CVPR 2026, arXiv 2026-04-23},
  urldate = {2026-10-03},
  eprint = {2604.21772},
  archivePrefix = {arXiv}
}

@misc{GE81,
  author = {Christina Baek and Yiding Jiang and Aditi Raghunathan and Zico Kolter},
  title = {{Agreement-on-the-Line: Predicting the Performance of Neural Networks under Distribution Shift}},
  year = {2022},
  url = {https://arxiv.org/abs/2206.13089},
  note = {GE81: NeurIPS 2022},
  urldate = {2026-10-03},
  eprint = {2206.13089},
  archivePrefix = {arXiv}
}

@misc{GE82,
  author = {Yiding Jiang and Vaishnavh Nagarajan and Christina Baek and J. Zico Kolter},
  title = {{Assessing Generalization of SGD via Disagreement}},
  year = {2021},
  url = {https://arxiv.org/abs/2106.13799},
  note = {GE82: arXiv 2021; ICLR 2022（公開PDF表紙で確認。掲載題名はAssessing Generalization via Disagreement）},
  urldate = {2026-10-03},
  eprint = {2106.13799},
  archivePrefix = {arXiv}
}

@misc{GE83,
  author = {Kamalika Chaudhuri and David Lopez-Paz},
  title = {{Unified Uncertainty Calibration}},
  year = {2023},
  url = {https://arxiv.org/abs/2310.01202},
  note = {GE83: arXiv preprint（2024改訂版確認）},
  urldate = {2026-10-03},
  eprint = {2310.01202},
  archivePrefix = {arXiv}
}

@misc{GE84,
  author = {Ishaan Gulrajani and Tatsunori Hashimoto},
  title = {{Identifiability Conditions for Domain Adaptation}},
  year = {2022},
  url = {https://proceedings.mlr.press/v162/gulrajani22a.html},
  note = {GE84: ICML 2022},
  urldate = {2026-10-03}
}

@misc{GE85,
  author = {Nanyang Ye and Kaican Li and Haoyue Bai and Runpeng Yu and Lanqing Hong and Fengwei Zhou and Zhenguo Li and Jun Zhu},
  title = {{OoD-Bench: Quantifying and Understanding Two Dimensions of Out-of-Distribution Generalization}},
  year = {2021},
  url = {https://arxiv.org/abs/2106.03721},
  note = {GE85: arXiv 2021; CVPR 2022},
  urldate = {2026-10-03},
  eprint = {2106.03721},
  archivePrefix = {arXiv}
}

@misc{GE86,
  author = {Robert Geirhos and Jörn-Henrik Jacobsen and Claudio Michaelis and Richard Zemel and Wieland Brendel and Matthias Bethge and Felix A. Wichmann},
  title = {{Shortcut Learning in Deep Neural Networks}},
  year = {2020},
  url = {https://arxiv.org/abs/2004.07780},
  note = {GE86: Nature Machine Intelligence 2020},
  urldate = {2026-10-03},
  eprint = {2004.07780},
  archivePrefix = {arXiv}
}

@misc{GE087,
  author = {Xingchao Peng and Ben Usman and Neela Kaushik and Judy Hoffman and Dequan Wang and Kate Saenko},
  title = {{VisDA: The Visual Domain Adaptation Challenge}},
  year = {2017},
  url = {https://arxiv.org/abs/1710.06924},
  note = {GE087: arXiv preprint, 2017},
  urldate = {2026-10-03},
  eprint = {1710.06924},
  archivePrefix = {arXiv}
}

@misc{GE088,
  author = {Anders Andreassen and Yasaman Bahri and Behnam Neyshabur and Rebecca Roelofs},
  title = {{The Evolution of Out-of-Distribution Robustness Throughout Fine-Tuning}},
  year = {2021},
  url = {https://arxiv.org/abs/2106.15831},
  note = {GE088: arXiv preprint, 2021（正式掲載先未照合）},
  urldate = {2026-10-03},
  eprint = {2106.15831},
  archivePrefix = {arXiv}
}

@misc{GE089,
  author = {Haotian Ye and Chuanlong Xie and Tianle Cai and Ruichen Li and Zhenguo Li and Liwei Wang},
  title = {{Towards a Theoretical Framework of Out-of-Distribution Generalization}},
  year = {2021},
  url = {https://arxiv.org/abs/2106.04496},
  note = {GE089: arXiv preprint, 2021（正式掲載先未照合）},
  urldate = {2026-10-03},
  eprint = {2106.04496},
  archivePrefix = {arXiv}
}

@misc{GE090,
  author = {Alexander Immer and Matthias Bauer and Vincent Fortuin and Gunnar Rätsch and Mohammad Emtiyaz Khan},
  title = {{Scalable Marginal Likelihood Estimation for Model Selection in Deep Learning}},
  year = {2021},
  url = {https://arxiv.org/abs/2104.04975},
  note = {GE090: ICML 2021},
  urldate = {2026-10-03},
  eprint = {2104.04975},
  archivePrefix = {arXiv}
}

@misc{GE091,
  author = {Eliran Shabat and Lee Cohen and Yishay Mansour},
  title = {{Sample Complexity of Uniform Convergence for Multicalibration}},
  year = {2020},
  url = {https://arxiv.org/abs/2005.01757},
  note = {GE091: NeurIPS 2020},
  urldate = {2026-10-03},
  eprint = {2005.01757},
  archivePrefix = {arXiv}
}

@misc{GE092,
  author = {Adarsh Subbaswamy and Peter Schulam and Suchi Saria},
  title = {{Preventing Failures Due to Dataset Shift: Learning Predictive Models That Transport}},
  year = {2018},
  url = {https://arxiv.org/abs/1812.04597},
  note = {GE092: arXiv 2018; AISTATS 2019},
  urldate = {2026-10-03},
  eprint = {1812.04597},
  archivePrefix = {arXiv}
}

@misc{GE093,
  author = {Amir Rahimi and Amirreza Shaban and Ching-An Cheng and Richard Hartley and Byron Boots},
  title = {{Intra Order-preserving Functions for Calibration of Multi-Class Neural Networks}},
  year = {2020},
  url = {https://arxiv.org/abs/2003.06820},
  note = {GE093: NeurIPS 2020},
  urldate = {2026-10-03},
  eprint = {2003.06820},
  archivePrefix = {arXiv}
}

@misc{GE094,
  author = {Martin Arjovsky},
  title = {{Out of Distribution Generalization in Machine Learning}},
  year = {2021},
  url = {https://arxiv.org/abs/2103.02667},
  note = {GE094: 博士論文のarXiv公開版, 2021},
  urldate = {2026-10-03},
  eprint = {2103.02667},
  archivePrefix = {arXiv}
}

@misc{GE095,
  author = {Marco Federici and Ryota Tomioka and Patrick Forré},
  title = {{An Information-theoretic Approach to Distribution Shifts}},
  year = {2021},
  url = {https://arxiv.org/abs/2106.03783},
  note = {GE095: arXiv preprint, 2021（正式掲載先未照合）},
  urldate = {2026-10-03},
  eprint = {2106.03783},
  archivePrefix = {arXiv}
}

@misc{GE096,
  author = {Dinghuai Zhang and Kartik Ahuja and Yilun Xu and Yisen Wang and Aaron Courville},
  title = {{Can Subnetwork Structure be the Key to Out-of-Distribution Generalization?}},
  year = {2021},
  url = {https://arxiv.org/abs/2106.02890},
  note = {GE096: ICML 2021},
  urldate = {2026-10-03},
  eprint = {2106.02890},
  archivePrefix = {arXiv}
}

@misc{GE097,
  author = {Xu Ji and Razvan Pascanu and Devon Hjelm and Balaji Lakshminarayanan and Andrea Vedaldi},
  title = {{Test Sample Accuracy Scales with Training Sample Density in Neural Networks}},
  year = {2021},
  url = {https://arxiv.org/abs/2106.08365},
  note = {GE097: arXiv 2021; CoLLAs 2022},
  urldate = {2026-10-03},
  eprint = {2106.08365},
  archivePrefix = {arXiv}
}

@misc{GE098,
  author = {Yisen Wang and Xingjun Ma and Zaiyi Chen and Yuan Luo and Jinfeng Yi and James Bailey},
  title = {{Symmetric Cross Entropy for Robust Learning with Noisy Labels}},
  year = {2019},
  url = {https://arxiv.org/abs/1908.06112},
  note = {GE098: ICCV 2019},
  urldate = {2026-10-03},
  eprint = {1908.06112},
  archivePrefix = {arXiv}
}

@misc{GE099,
  author = {Marvin Zhang and Henrik Marklund and Nikita Dhawan and Abhishek Gupta and Sergey Levine and Chelsea Finn},
  title = {{Adaptive Risk Minimization: Learning to Adapt to Domain Shift}},
  year = {2020},
  url = {https://arxiv.org/abs/2007.02931},
  note = {GE099: arXiv 2020; NeurIPS 2021},
  urldate = {2026-10-03},
  eprint = {2007.02931},
  archivePrefix = {arXiv}
}

@misc{GE100,
  author = {Kai Xiao and Logan Engstrom and Andrew Ilyas and Aleksander Madry},
  title = {{Noise or Signal: The Role of Image Backgrounds in Object Recognition}},
  year = {2020},
  url = {https://arxiv.org/abs/2006.09994},
  note = {GE100: arXiv preprint, 2020（正式掲載先未照合）},
  urldate = {2026-10-03},
  eprint = {2006.09994},
  archivePrefix = {arXiv}
}

@misc{GE101,
  author = {Yifei Ming and Hang Yin and Yixuan Li},
  title = {{On the Impact of Spurious Correlation for Out-of-distribution Detection}},
  year = {2021},
  url = {https://arxiv.org/abs/2109.05642},
  note = {GE101: arXiv 2021; AAAI 2022},
  urldate = {2026-10-03},
  eprint = {2109.05642},
  archivePrefix = {arXiv}
}

@misc{GE102,
  author = {Marc Khoury},
  title = {{Adaptive versus Standard Descent Methods and Robustness Against Adversarial Examples}},
  year = {2019},
  url = {https://arxiv.org/abs/1911.03784},
  note = {GE102: arXiv preprint, 2019（2020改訂）},
  urldate = {2026-10-03},
  eprint = {1911.03784},
  archivePrefix = {arXiv}
}

@misc{GE103,
  author = {Nader Asadi and Amir M. Sarfi and Mehrdad Hosseinzadeh and Zahra Karimpour and Mahdi Eftekhari},
  title = {{Towards Shape Biased Unsupervised Representation Learning for Domain Generalization}},
  year = {2019},
  url = {https://arxiv.org/abs/1909.08245},
  note = {GE103: arXiv preprint, 2019（2020改訂、正式掲載先未照合）},
  urldate = {2026-10-03},
  eprint = {1909.08245},
  archivePrefix = {arXiv}
}

@misc{GE104,
  author = {Rui Hu and Jitao Sang and Jinqiang Wang and Rui Hu and Chaoquan Jiang},
  title = {{Understanding and Testing Generalization of Deep Networks on Out-of-Distribution Data}},
  year = {2021},
  url = {https://arxiv.org/abs/2111.09190},
  note = {GE104: arXiv preprint, 2021},
  urldate = {2026-10-03},
  eprint = {2111.09190},
  archivePrefix = {arXiv}
}

@misc{GE105,
  author = {Srinadh Bhojanapalli and Ayan Chakrabarti and Daniel Glasner and Daliang Li and Thomas Unterthiner and Andreas Veit},
  title = {{Understanding Robustness of Transformers for Image Classification}},
  year = {2021},
  url = {https://arxiv.org/abs/2103.14586},
  note = {GE105: ICCV 2021},
  urldate = {2026-10-03},
  eprint = {2103.14586},
  archivePrefix = {arXiv}
}

@misc{GE106,
  author = {Dan Hendrycks and Xiaoyuan Liu and Eric Wallace and Adam Dziedzic and Rishabh Krishnan and Dawn Song},
  title = {{Pretrained Transformers Improve Out-of-Distribution Robustness}},
  year = {2020},
  url = {https://arxiv.org/abs/2004.06100},
  note = {GE106: ACL 2020},
  urldate = {2026-10-03},
  eprint = {2004.06100},
  archivePrefix = {arXiv}
}

@misc{GE107,
  author = {Zachary Nado and Neil Band and Mark Collier and Josip Djolonga and Michael W. Dusenberry and Sebastian Farquhar and Qixuan Feng and Angelos Filos and Marton Havasi and Rodolphe Jenatton and Ghassen Jerfel and Jeremiah Liu and Zelda Mariet and Jeremy Nixon and Shreyas Padhy and Jie Ren and Tim G. J. Rudner and Faris Sbahi and Yeming Wen and Florian Wenzel and Kevin Murphy and D. Sculley and Balaji Lakshminarayanan and Jasper Snoek and Yarin Gal and Dustin Tran},
  title = {{Uncertainty Baselines: Benchmarks for Uncertainty \& Robustness in Deep Learning}},
  year = {2021},
  url = {https://arxiv.org/abs/2106.04015},
  note = {GE107: arXiv preprint, 2021（2022改訂）},
  urldate = {2026-10-03},
  eprint = {2106.04015},
  archivePrefix = {arXiv}
}

@misc{GE108,
  author = {James Diffenderfer and Brian R. Bartoldson and Shreya Chaganti and Jize Zhang and Bhavya Kailkhura},
  title = {{A Winning Hand: Compressing Deep Networks Can Improve Out-Of-Distribution Robustness}},
  year = {2021},
  url = {https://arxiv.org/abs/2106.09129},
  note = {GE108: arXiv preprint, 2021（正式掲載先未照合）},
  urldate = {2026-10-03},
  eprint = {2106.09129},
  archivePrefix = {arXiv}
}

@misc{GE109,
  author = {Harshay Shah and Kaustav Tamuly and Aditi Raghunathan and Prateek Jain and Praneeth Netrapalli},
  title = {{The Pitfalls of Simplicity Bias in Neural Networks}},
  year = {2020},
  url = {https://arxiv.org/abs/2006.07710},
  note = {GE109: NeurIPS 2020},
  urldate = {2026-10-03},
  eprint = {2006.07710},
  archivePrefix = {arXiv}
}

@misc{GE110,
  author = {Muhammad Ghifary and W. Bastiaan Kleijn and Mengjie Zhang and David Balduzzi},
  title = {{Domain Generalization for Object Recognition with Multi-task Autoencoders}},
  year = {2015},
  url = {https://arxiv.org/abs/1508.07680},
  note = {GE110: ICCV 2015},
  urldate = {2026-10-03},
  eprint = {1508.07680},
  archivePrefix = {arXiv}
}

@misc{GE111,
  author = {Da Li and Yongxin Yang and Yi-Zhe Song and Timothy M. Hospedales},
  title = {{Deeper, Broader and Artier Domain Generalization}},
  year = {2017},
  url = {https://arxiv.org/abs/1710.03077},
  note = {GE111: ICCV 2017},
  urldate = {2026-10-03},
  eprint = {1710.03077},
  archivePrefix = {arXiv}
}

@misc{GE112,
  author = {Sara Beery and Grant van Horn and Pietro Perona},
  title = {{Recognition in Terra Incognita}},
  year = {2018},
  url = {https://arxiv.org/abs/1807.04975},
  note = {GE112: ECCV 2018},
  urldate = {2026-10-03},
  eprint = {1807.04975},
  archivePrefix = {arXiv}
}

@misc{GE113,
  author = {Xingchao Peng and Qinxun Bai and Xide Xia and Zijun Huang and Kate Saenko and Bo Wang},
  title = {{Moment Matching for Multi-Source Domain Adaptation}},
  year = {2018},
  url = {https://arxiv.org/abs/1812.01754},
  note = {GE113: arXiv 2018; ICCV 2019},
  urldate = {2026-10-03},
  eprint = {1812.01754},
  archivePrefix = {arXiv}
}

@misc{GE114,
  author = {Robert Geirhos and Patricia Rubisch and Claudio Michaelis and Matthias Bethge and Felix A. Wichmann and Wieland Brendel},
  title = {{ImageNet-trained CNNs are biased towards texture; increasing shape bias improves accuracy and robustness}},
  year = {2018},
  url = {https://arxiv.org/abs/1811.12231},
  note = {GE114: arXiv 2018; ICLR 2019},
  urldate = {2026-10-03},
  eprint = {1811.12231},
  archivePrefix = {arXiv}
}

@misc{GE115,
  author = {Pang Wei Koh and Thao Nguyen and Yew Siang Tang and Stephen Mussmann and Emma Pierson and Been Kim and Percy Liang},
  title = {{Concept Bottleneck Models}},
  year = {2020},
  url = {https://arxiv.org/abs/2007.04612},
  note = {GE115: ICML 2020},
  urldate = {2026-10-03},
  eprint = {2007.04612},
  archivePrefix = {arXiv}
}

@misc{GE116,
  author = {Vaishaal Shankar and Achal Dave and Rebecca Roelofs and Deva Ramanan and Benjamin Recht and Ludwig Schmidt},
  title = {{Do Image Classifiers Generalize Across Time?}},
  year = {2019},
  url = {https://arxiv.org/abs/1906.02168},
  note = {GE116: arXiv preprint, 2019（正式掲載先未照合）},
  urldate = {2026-10-03},
  eprint = {1906.02168},
  archivePrefix = {arXiv}
}

@misc{GE117,
  author = {Eleni Triantafillou and Tyler Zhu and Vincent Dumoulin and Pascal Lamblin and Utku Evci and Kelvin Xu and Ross Goroshin and Carles Gelada and Kevin Swersky and Pierre-Antoine Manzagol and Hugo Larochelle},
  title = {{Meta-Dataset: A Dataset of Datasets for Learning to Learn from Few Examples}},
  year = {2019},
  url = {https://arxiv.org/abs/1903.03096},
  note = {GE117: arXiv 2019; ICLR 2020},
  urldate = {2026-10-03},
  eprint = {1903.03096},
  archivePrefix = {arXiv}
}

@misc{GE118,
  author = {Xiaohua Zhai and Joan Puigcerver and Alexander Kolesnikov and Pierre Ruyssen and Carlos Riquelme and Mario Lucic and Josip Djolonga and Andre Susano Pinto and Maxim Neumann and Alexey Dosovitskiy and Lucas Beyer and Olivier Bachem and Michael Tschannen and Marcin Michalski and Olivier Bousquet and Sylvain Gelly and Neil Houlsby},
  title = {{A Large-scale Study of Representation Learning with the Visual Task Adaptation Benchmark}},
  year = {2019},
  url = {https://arxiv.org/abs/1910.04867},
  note = {GE118: arXiv preprint, 2019（2020改訂、正式掲載先未照合）},
  urldate = {2026-10-03},
  eprint = {1910.04867},
  archivePrefix = {arXiv}
}

@misc{GE119,
  author = {Aishwarya Agrawal and Dhruv Batra and Devi Parikh and Aniruddha Kembhavi},
  title = {{Don't Just Assume; Look and Answer: Overcoming Priors for Visual Question Answering}},
  year = {2017},
  url = {https://arxiv.org/abs/1712.00377},
  note = {GE119: arXiv 2017; CVPR 2018},
  urldate = {2026-10-03},
  eprint = {1712.00377},
  archivePrefix = {arXiv}
}

@misc{GE120,
  author = {Yixin Nie and Adina Williams and Emily Dinan and Mohit Bansal and Jason Weston and Douwe Kiela},
  title = {{Adversarial NLI: A New Benchmark for Natural Language Understanding}},
  year = {2019},
  url = {https://arxiv.org/abs/1910.14599},
  note = {GE120: arXiv 2019; ACL 2020},
  urldate = {2026-10-03},
  eprint = {1910.14599},
  archivePrefix = {arXiv}
}

@misc{GE121,
  author = {Dan Hendrycks and Kevin Zhao and Steven Basart and Jacob Steinhardt and Dawn Song},
  title = {{Natural Adversarial Examples}},
  year = {2019},
  url = {https://arxiv.org/abs/1907.07174},
  note = {GE121: arXiv 2019; CVPR 2021},
  urldate = {2026-10-03},
  eprint = {1907.07174},
  archivePrefix = {arXiv}
}

@misc{GE122,
  author = {Olivia Wiles and Sven Gowal and Florian Stimberg and Sylvestre Alvise-Rebuffi and Ira Ktena and Krishnamurthy Dvijotham and Taylan Cemgil},
  title = {{A Fine-Grained Analysis on Distribution Shift}},
  year = {2021},
  url = {https://arxiv.org/abs/2110.11328},
  note = {GE122: arXiv 2021; ICLR 2022},
  urldate = {2026-10-03},
  eprint = {2110.11328},
  archivePrefix = {arXiv}
}

@misc{GE123,
  author = {Tian Li and Ahmad Beirami and Maziar Sanjabi and Virginia Smith},
  title = {{Tilted Empirical Risk Minimization}},
  year = {2020},
  url = {https://arxiv.org/abs/2007.01162},
  note = {GE123: arXiv 2020; ICLR 2021},
  urldate = {2026-10-03},
  eprint = {2007.01162},
  archivePrefix = {arXiv}
}

@misc{GE124,
  author = {Wang Lu and Jindong Wang and Haoliang Li and Yiqiang Chen and Xing Xie},
  title = {{Domain-invariant Feature Exploration for Domain Generalization}},
  year = {2022},
  url = {https://arxiv.org/abs/2207.12020},
  note = {GE124: TMLR 2022},
  urldate = {2026-10-03},
  eprint = {2207.12020},
  archivePrefix = {arXiv}
}

@misc{GE125,
  author = {Andreas Kirsch and Yarin Gal},
  title = {{A Note on "Assessing Generalization of SGD via Disagreement"}},
  year = {2022},
  url = {https://arxiv.org/abs/2202.01851},
  note = {GE125: arXiv 2022; TMLR 2022（公開PDF表紙で確認）},
  urldate = {2026-10-03},
  eprint = {2202.01851},
  archivePrefix = {arXiv}
}

@misc{GE126,
  author = {Utku Evci and Vincent Dumoulin and Hugo Larochelle and Michael C. Mozer},
  title = {{Head2Toe: Utilizing Intermediate Representations for Better Transfer Learning}},
  year = {2022},
  url = {https://arxiv.org/abs/2201.03529},
  note = {GE126: ICML 2022},
  urldate = {2026-10-03},
  eprint = {2201.03529},
  archivePrefix = {arXiv}
}

@misc{GE127,
  author = {Nathan Ng and Neha Hulkund and Kyunghyun Cho and Marzyeh Ghassemi},
  title = {{Predicting Out-of-Domain Generalization with Neighborhood Invariance}},
  year = {2022},
  url = {https://arxiv.org/abs/2207.02093},
  note = {GE127: arXiv 2022; TMLR 2023（公開PDF表紙で確認）},
  urldate = {2026-10-03},
  eprint = {2207.02093},
  archivePrefix = {arXiv}
}

@misc{GE128,
  author = {Jingling Li and Mozhi Zhang and Keyulu Xu and John P. Dickerson and Jimmy Ba},
  title = {{How Does a Neural Network's Architecture Impact Its Robustness to Noisy Labels?}},
  year = {2020},
  url = {https://arxiv.org/abs/2012.12896},
  note = {GE128: arXiv 2020; NeurIPS 2021},
  urldate = {2026-10-03},
  eprint = {2012.12896},
  archivePrefix = {arXiv}
}

@misc{GE129,
  author = {Masanori Koyama and Shoichiro Yamaguchi},
  title = {{When is invariance useful in an Out-of-Distribution Generalization problem ?}},
  year = {2020},
  url = {https://arxiv.org/abs/2008.01883},
  note = {GE129: arXiv preprint, 2020（2021改訂）},
  urldate = {2026-10-03},
  eprint = {2008.01883},
  archivePrefix = {arXiv}
}

@misc{GE130,
  author = {Leo Kozachkov and Patrick M. Wensing and Jean-Jacques Slotine},
  title = {{Generalization in Supervised Learning Through Riemannian Contraction}},
  year = {2022},
  url = {https://arxiv.org/abs/2201.06656},
  note = {GE130: arXiv 2022; 発展版はGeneralization as Dynamical Robustness–The Role of Riemannian Contraction in Supervised Learning, TMLR 2023（書誌照合）},
  urldate = {2026-10-03},
  eprint = {2201.06656},
  archivePrefix = {arXiv}
}

@misc{GE131,
  author = {Daniel C. Castro and Ian Walker and Ben Glocker},
  title = {{Causality matters in medical imaging}},
  year = {2019},
  url = {https://arxiv.org/abs/1912.08142},
  note = {GE131: arXiv 2019; Nature Communications 11:3673, 2020},
  urldate = {2026-10-03},
  eprint = {1912.08142},
  archivePrefix = {arXiv}
}

@misc{GE132,
  author = {Mateusz Michalkiewicz and Masoud Faraki and Xiang Yu and Manmohan Chandraker and Mahsa Baktashmotlagh},
  title = {{Domain Generalization Guided by Gradient Signal to Noise Ratio of Parameters}},
  year = {2023},
  url = {https://arxiv.org/abs/2310.07361},
  note = {GE132: ICCV 2023},
  urldate = {2026-10-03},
  eprint = {2310.07361},
  archivePrefix = {arXiv}
}

@misc{GE133,
  author = {Hemanth Venkateswara and Jose Eusebio and Shayok Chakraborty and Sethuraman Panchanathan},
  title = {{Deep Hashing Network for Unsupervised Domain Adaptation}},
  year = {2017},
  url = {https://arxiv.org/abs/1706.07522},
  note = {GE133: CVPR 2017},
  urldate = {2026-10-03},
  eprint = {1706.07522},
  archivePrefix = {arXiv}
}

@misc{GE134,
  author = {Yinpeng Dong and Qi-An Fu and Xiao Yang and Tianyu Pang and Hang Su and Zihao Xiao and Jun Zhu},
  title = {{Benchmarking Adversarial Robustness}},
  year = {2019},
  url = {https://arxiv.org/abs/1912.11852},
  note = {GE134: arXiv 2019; CVPR 2020掲載題名は「Benchmarking Adversarial Robustness on Image Classification」},
  urldate = {2026-10-03},
  eprint = {1912.11852},
  archivePrefix = {arXiv}
}

@misc{GE135,
  author = {Sumin Cho and Dongwon Kim and Kwangsu Kim},
  title = {{One-Step Generalization Ratio Guided Optimization for Domain Generalization}},
  year = {2025},
  url = {https://proceedings.mlr.press/v267/cho25c.html},
  note = {GE135: ICML 2025; arXivへの登録は2026-06-15},
  urldate = {2026-10-03}
}

@misc{GE136,
  author = {Dan Hendrycks and Steven Basart and Norman Mu and Saurav Kadavath and Frank Wang and Evan Dorundo and Rahul Desai and Tyler Zhu and Samyak Parajuli and Mike Guo and Dawn Song and Jacob Steinhardt and Justin Gilmer},
  title = {{The Many Faces of Robustness: A Critical Analysis of Out-of-Distribution Generalization}},
  year = {2020},
  url = {https://arxiv.org/abs/2006.16241},
  note = {GE136: arXiv 2020; ICCV 2021 / OA038: ICCV 2021},
  urldate = {2026-10-03},
  eprint = {2006.16241},
  archivePrefix = {arXiv}
}

@misc{GE137,
  author = {Ananya Kumar and Tengyu Ma and Percy Liang and Aditi Raghunathan},
  title = {{Calibrated ensembles can mitigate accuracy tradeoffs under distribution shift}},
  year = {2022},
  url = {https://proceedings.mlr.press/v180/kumar22a.html},
  note = {GE137: UAI 2022},
  urldate = {2026-10-03}
}

@misc{GE138,
  author = {Joaquin Quiñonero-Candela (編) and Masashi Sugiyama (編) and Anton Schwaighofer (編) and Neil D. Lawrence (編)},
  title = {{Dataset Shift in Machine Learning}},
  year = {2008},
  url = {https://mitpress.mit.edu/9780262545877/dataset-shift-in-machine-learning/},
  note = {GE138: MIT Press（出版社はhardcover発売2008-12-12、paperback2022-06-07と表示。一般的な引用年は2009）},
  urldate = {2026-10-03}
}

@misc{GE139,
  author = {Mehdi Ataei and Murat Erdogdu and Sedef Akinli Kocak and Shai Ben-David and Shems Saleh},
  title = {{Understanding Dataset Shift and Potential Remedies}},
  year = {2021},
  url = {https://vectorinstitute.ai/wp-content/uploads/2021/08/ds_project_report_final_august9.pdf},
  note = {GE139: Vector Institute Industry Collaborative Project Technical Report, 2021},
  urldate = {2026-10-03}
}

@misc{GE140,
  author = {Reva Schwartz and Apostol Vassilev and Kristen Greene and Lori Perine and Andrew Burt and Patrick Hall},
  title = {{Towards a Standard for Identifying and Managing Bias in Artificial Intelligence}},
  year = {2022},
  url = {https://doi.org/10.6028/NIST.SP.1270},
  note = {GE140: NIST Special Publication 1270, March 2022},
  urldate = {2026-10-03}
}

@misc{GE141,
  author = {Qi Qi and Zhishuai Guo and Yi Xu and Rong Jin and Tianbao Yang},
  title = {{An Online Method for A Class of Distributionally Robust Optimization with Non-convex Objectives}},
  year = {2021},
  url = {https://proceedings.neurips.cc/paper/2021/hash/533fa796b43291fc61a9e812a50c3fb6-Abstract.html},
  note = {GE141: NeurIPS 2021},
  urldate = {2026-10-03}
}

@misc{GE142,
  author = {Boyan Gao and Henry Gouk and Yongxin Yang and Timothy Hospedales},
  title = {{Loss Function Learning for Domain Generalization by Implicit Gradient}},
  year = {2022},
  url = {https://proceedings.mlr.press/v162/gao22b.html},
  note = {GE142: ICML 2022},
  urldate = {2026-10-03}
}

@misc{GE143,
  author = {Ramakrishna Vedantam and David Lopez-Paz and David J. Schwab},
  title = {{An Empirical Investigation of Domain Generalization with Empirical Risk Minimizers}},
  year = {2021},
  url = {https://proceedings.neurips.cc/paper_files/paper/2021/hash/ecf9902e0f61677c8de25ae60b654669-Abstract.html},
  note = {GE143: NeurIPS 2021},
  urldate = {2026-10-03}
}

@misc{GE144,
  author = {Yao-Yuan Yang and Cyrus Rashtchian and Hongyang Zhang and Ruslan Salakhutdinov and Kamalika Chaudhuri},
  title = {{A Closer Look at Accuracy vs. Robustness}},
  year = {2020},
  url = {https://proceedings.neurips.cc/paper/2020/hash/61d77652c97ef636343742fc3dcf3ba9-Abstract.html},
  note = {GE144: NeurIPS 2020},
  urldate = {2026-10-03}
}

@misc{GE145,
  author = {Chen Fang and Ye Xu and Daniel N. Rockmore},
  title = {{Unbiased Metric Learning: On the Utilization of Multiple Datasets and Web Images for Softening Bias}},
  year = {2013},
  url = {https://www.cv-foundation.org/openaccess/content_iccv_2013/papers/Fang_Unbiased_Metric_Learning_2013_ICCV_paper.pdf},
  note = {GE145: ICCV 2013},
  urldate = {2026-10-03}
}

@misc{GE146,
  author = {Sara Beery and Guanhang Wu and Trevor Edwards and Filip Pavetic and Bo Majewski and Shreyasee Mukherjee and Stanley Chan and John Morgan and Vivek Rathod and Jonathan Huang},
  title = {{The Auto Arborist Dataset: A Large-Scale Benchmark for Multiview Urban Forest Monitoring Under Domain Shift}},
  year = {2022},
  url = {https://openaccess.thecvf.com/content/CVPR2022/html/Beery_The_Auto_Arborist_Dataset_A_Large-Scale_Benchmark_for_Multiview_Urban_CVPR_2022_paper.html},
  note = {GE146: CVPR 2022（正式書誌と著者の公式データセット説明を確認。会議PDF本文未取得）},
  urldate = {2026-10-03}
}

@misc{GE147,
  author = {Tobias Ringwald and Rainer Stiefelhagen},
  title = {{Adaptiope: A Modern Benchmark for Unsupervised Domain Adaptation}},
  year = {2021},
  url = {https://publikationen.bibliothek.kit.edu/1000142108},
  note = {GE147: WACV 2021, pp.101–110; DOI:10.1109/WACV48630.2021.00015},
  urldate = {2026-10-03}
}

@misc{FM001,
  author = {Dzmitry Bahdanau and others},
  title = {{Neural Machine Translation by Jointly Learning to Align and Translate}},
  year = {2014},
  url = {https://arxiv.org/abs/1409.0473},
  note = {FM001: ICLR 2015},
  urldate = {2026-10-03},
  eprint = {1409.0473},
  archivePrefix = {arXiv}
}

@misc{FM002,
  author = {Ashish Vaswani and others},
  title = {{Attention Is All You Need}},
  year = {2017},
  url = {https://arxiv.org/abs/1706.03762},
  note = {FM002: NeurIPS 2017},
  urldate = {2026-10-03},
  eprint = {1706.03762},
  archivePrefix = {arXiv}
}

@misc{FM003,
  author = {Jacob Devlin and others},
  title = {{BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding}},
  year = {2018},
  url = {https://arxiv.org/abs/1810.04805},
  note = {FM003: arXiv preprint（会議版未再照合）},
  urldate = {2026-10-03},
  eprint = {1810.04805},
  archivePrefix = {arXiv}
}

@misc{FM004,
  author = {Tom B. Brown and others},
  title = {{Language Models are Few-Shot Learners}},
  year = {2020},
  url = {https://arxiv.org/abs/2005.14165},
  note = {FM004: NeurIPS 2020},
  urldate = {2026-10-03},
  eprint = {2005.14165},
  archivePrefix = {arXiv}
}

@misc{FM005,
  author = {Alexey Dosovitskiy and others},
  title = {{An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale}},
  year = {2020},
  url = {https://arxiv.org/abs/2010.11929},
  note = {FM005: ICLR 2021},
  urldate = {2026-10-03},
  eprint = {2010.11929},
  archivePrefix = {arXiv}
}

@misc{FM006,
  author = {Jianlin Su and others},
  title = {{RoFormer: Enhanced Transformer with Rotary Position Embedding}},
  year = {2021},
  url = {https://arxiv.org/abs/2104.09864},
  note = {FM006: arXiv preprint（出版版未再照合）},
  urldate = {2026-10-03},
  eprint = {2104.09864},
  archivePrefix = {arXiv}
}

@misc{FM007,
  author = {Noam Shazeer},
  title = {{Fast Transformer Decoding: One Write-Head is All You Need}},
  year = {2019},
  url = {https://arxiv.org/abs/1911.02150},
  note = {FM007: arXiv preprint},
  urldate = {2026-10-03},
  eprint = {1911.02150},
  archivePrefix = {arXiv}
}

@misc{FM008,
  author = {Joshua Ainslie and others},
  title = {{GQA: Training Generalized Multi-Query Transformer Models from Multi-Head Checkpoints}},
  year = {2023},
  url = {https://arxiv.org/abs/2305.13245},
  note = {FM008: EMNLP 2023},
  urldate = {2026-10-03},
  eprint = {2305.13245},
  archivePrefix = {arXiv}
}

@misc{FM009,
  author = {Krzysztof Choromanski and others},
  title = {{Rethinking Attention with Performers}},
  year = {2020},
  url = {https://arxiv.org/abs/2009.14794},
  note = {FM009: ICLR 2021},
  urldate = {2026-10-03},
  eprint = {2009.14794},
  archivePrefix = {arXiv}
}

@misc{FM010,
  author = {Tri Dao and others},
  title = {{FlashAttention: Fast and Memory-Efficient Exact Attention with IO-Awareness}},
  year = {2022},
  url = {https://arxiv.org/abs/2205.14135},
  note = {FM010: arXiv preprint（会議版未再照合） / SY009: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2205.14135},
  archivePrefix = {arXiv}
}

@misc{FM011,
  author = {Tri Dao},
  title = {{FlashAttention-2: Faster Attention with Better Parallelism and Work Partitioning}},
  year = {2023},
  url = {https://arxiv.org/abs/2307.08691},
  note = {FM011: arXiv preprint（会議版未再照合）},
  urldate = {2026-10-03},
  eprint = {2307.08691},
  archivePrefix = {arXiv}
}

@misc{FM012,
  author = {Jay Shah and others},
  title = {{FlashAttention-3: Fast and Accurate Attention with Asynchrony and Low-precision}},
  year = {2024},
  url = {https://arxiv.org/abs/2407.08608},
  note = {FM012: arXiv preprint},
  urldate = {2026-10-03},
  eprint = {2407.08608},
  archivePrefix = {arXiv}
}

@misc{FM013,
  author = {Ted Zadouri and others},
  title = {{FlashAttention-4: Algorithm and Kernel Pipelining Co-Design for Asymmetric Hardware Scaling}},
  year = {2026},
  url = {https://arxiv.org/abs/2603.05451},
  note = {FM013: arXiv preprint},
  urldate = {2026-10-03},
  eprint = {2603.05451},
  archivePrefix = {arXiv}
}

@misc{FM014,
  author = {Albert Gu and Tri Dao},
  title = {{Mamba: Linear-Time Sequence Modeling with Selective State Spaces}},
  year = {2023},
  url = {https://arxiv.org/abs/2312.00752},
  note = {FM014: arXiv preprint（会議版未再照合）},
  urldate = {2026-10-03},
  eprint = {2312.00752},
  archivePrefix = {arXiv}
}

@misc{FM015,
  author = {Tri Dao and Albert Gu},
  title = {{Transformers are SSMs: Generalized Models and Efficient Algorithms Through Structured State Space Duality}},
  year = {2024},
  url = {https://arxiv.org/abs/2405.21060},
  note = {FM015: ICML 2024},
  urldate = {2026-10-03},
  eprint = {2405.21060},
  archivePrefix = {arXiv}
}

@misc{FM016,
  author = {DeepSeek-AI and others},
  title = {{DeepSeek-V3 Technical Report}},
  year = {2024},
  url = {https://arxiv.org/abs/2412.19437},
  note = {FM016: arXiv technical report / SY014: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2412.19437},
  archivePrefix = {arXiv}
}

@misc{FM017,
  author = {Joel Hestness and others},
  title = {{Deep Learning Scaling is Predictable, Empirically}},
  year = {2017},
  url = {https://arxiv.org/abs/1712.00409},
  note = {FM017: arXiv preprint},
  urldate = {2026-10-03},
  eprint = {1712.00409},
  archivePrefix = {arXiv}
}

@misc{FM018,
  author = {Jared Kaplan and others},
  title = {{Scaling Laws for Neural Language Models}},
  year = {2020},
  url = {https://arxiv.org/abs/2001.08361},
  note = {FM018: arXiv preprint},
  urldate = {2026-10-03},
  eprint = {2001.08361},
  archivePrefix = {arXiv}
}

@misc{FM019,
  author = {Jordan Hoffmann and others},
  title = {{Training Compute-Optimal Large Language Models}},
  year = {2022},
  url = {https://arxiv.org/abs/2203.15556},
  note = {FM019: arXiv preprint（会議版未再照合）},
  urldate = {2026-10-03},
  eprint = {2203.15556},
  archivePrefix = {arXiv}
}

@misc{FM020,
  author = {Xiaohua Zhai and others},
  title = {{Scaling Vision Transformers}},
  year = {2021},
  url = {https://arxiv.org/abs/2106.04560},
  note = {FM020: CVPR 2022},
  urldate = {2026-10-03},
  eprint = {2106.04560},
  archivePrefix = {arXiv}
}

@misc{FM021,
  author = {Yasaman Bahri and others},
  title = {{Explaining Neural Scaling Laws}},
  year = {2021},
  url = {https://arxiv.org/abs/2102.06701},
  note = {FM021: PNAS 2024},
  urldate = {2026-10-03},
  eprint = {2102.06701},
  archivePrefix = {arXiv}
}

@misc{FM022,
  author = {Niklas Muennighoff and others},
  title = {{Scaling Data-Constrained Language Models}},
  year = {2023},
  url = {https://arxiv.org/abs/2305.16264},
  note = {FM022: arXiv preprint（会議版未再照合）},
  urldate = {2026-10-03},
  eprint = {2305.16264},
  archivePrefix = {arXiv}
}

@misc{FM023,
  author = {Pablo Villalobos and others},
  title = {{Will we run out of data? Limits of LLM scaling based on human-generated data}},
  year = {2022},
  url = {https://arxiv.org/abs/2211.04325},
  note = {FM023: arXiv preprint（会議版未再照合）},
  urldate = {2026-10-03},
  eprint = {2211.04325},
  archivePrefix = {arXiv}
}

@misc{FM024,
  author = {Rylan Schaeffer and others},
  title = {{Are Emergent Abilities of Large Language Models a Mirage?}},
  year = {2023},
  url = {https://arxiv.org/abs/2304.15004},
  note = {FM024: arXiv preprint（会議版未再照合）},
  urldate = {2026-10-03},
  eprint = {2304.15004},
  archivePrefix = {arXiv}
}

@misc{FM025,
  author = {Song Bian and others},
  title = {{Scaling Inference-Efficient Language Models}},
  year = {2025},
  url = {https://arxiv.org/abs/2501.18107},
  note = {FM025: ICML 2025},
  urldate = {2026-10-03},
  eprint = {2501.18107},
  archivePrefix = {arXiv}
}

@misc{FM026,
  author = {Shayne Longpre and others},
  title = {{ATLAS: Adaptive Transfer Scaling Laws for Multilingual Pretraining, Finetuning, and Decoding the Curse of Multilinguality}},
  year = {2025},
  url = {https://arxiv.org/abs/2510.22037},
  note = {FM026: ICLR 2026（著者公式発表で確認）},
  urldate = {2026-10-03},
  eprint = {2510.22037},
  archivePrefix = {arXiv}
}

@misc{FM027,
  author = {Xinye Zhao and others},
  title = {{Not All Error Yields to Scale: Where Scaling Stops in Vision-Language Inference}},
  year = {2026},
  url = {https://arxiv.org/abs/2610.01640},
  note = {FM027: arXiv preprint},
  urldate = {2026-10-03},
  eprint = {2610.01640},
  archivePrefix = {arXiv}
}

@misc{FM028,
  author = {Tadas Baltrušaitis and others},
  title = {{Multimodal Machine Learning: A Survey and Taxonomy}},
  year = {2017},
  url = {https://arxiv.org/abs/1705.09406},
  note = {FM028: arXiv preprint（雑誌版未再照合）},
  urldate = {2026-10-03},
  eprint = {1705.09406},
  archivePrefix = {arXiv}
}

@misc{FM029,
  author = {Alec Radford and others},
  title = {{Learning Transferable Visual Models From Natural Language Supervision}},
  year = {2021},
  url = {https://arxiv.org/abs/2103.00020},
  note = {FM029: arXiv preprint（会議版未再照合）},
  urldate = {2026-10-03},
  eprint = {2103.00020},
  archivePrefix = {arXiv}
}

@misc{FM030,
  author = {Jean-Baptiste Alayrac and others},
  title = {{Flamingo: a Visual Language Model for Few-Shot Learning}},
  year = {2022},
  url = {https://arxiv.org/abs/2204.14198},
  note = {FM030: NeurIPS 2022},
  urldate = {2026-10-03},
  eprint = {2204.14198},
  archivePrefix = {arXiv}
}

@misc{FM031,
  author = {Junnan Li and others},
  title = {{BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models}},
  year = {2023},
  url = {https://arxiv.org/abs/2301.12597},
  note = {FM031: arXiv preprint（会議版未再照合）},
  urldate = {2026-10-03},
  eprint = {2301.12597},
  archivePrefix = {arXiv}
}

@misc{FM032,
  author = {Haotian Liu and others},
  title = {{Visual Instruction Tuning}},
  year = {2023},
  url = {https://arxiv.org/abs/2304.08485},
  note = {FM032: NeurIPS 2023},
  urldate = {2026-10-03},
  eprint = {2304.08485},
  archivePrefix = {arXiv}
}

@misc{FM033,
  author = {Rohit Girdhar and others},
  title = {{ImageBind: One Embedding Space To Bind Them All}},
  year = {2023},
  url = {https://arxiv.org/abs/2305.05665},
  note = {FM033: CVPR 2023},
  urldate = {2026-10-03},
  eprint = {2305.05665},
  archivePrefix = {arXiv}
}

@misc{FM034,
  author = {Shuai Bai and others},
  title = {{Qwen2.5-VL Technical Report}},
  year = {2025},
  url = {https://arxiv.org/abs/2502.13923},
  note = {FM034: arXiv technical report},
  urldate = {2026-10-03},
  eprint = {2502.13923},
  archivePrefix = {arXiv}
}

@misc{FM035,
  author = {Jin Xu and others},
  title = {{Qwen3-Omni Technical Report}},
  year = {2025},
  url = {https://arxiv.org/abs/2509.17765},
  note = {FM035: arXiv technical report},
  urldate = {2026-10-03},
  eprint = {2509.17765},
  archivePrefix = {arXiv}
}

@misc{FM036,
  author = {George Barnum and others},
  title = {{On the Benefits of Early Fusion in Multimodal Representation Learning}},
  year = {2020},
  url = {https://arxiv.org/abs/2011.07191},
  note = {FM036: arXiv preprint（ICLR採択は未確認）},
  urldate = {2026-10-03},
  eprint = {2011.07191},
  archivePrefix = {arXiv}
}

@misc{FM037,
  author = {Renjie Wu and others},
  title = {{Deep Multimodal Learning with Missing Modality: A Survey}},
  year = {2024},
  url = {https://arxiv.org/abs/2409.07825},
  note = {FM037: TMLR（arXiv更新版で採択記載、2026）},
  urldate = {2026-10-03},
  eprint = {2409.07825},
  archivePrefix = {arXiv}
}

@misc{FM038,
  author = {Pan Wang and others},
  title = {{From Models to Systems: A Comprehensive Survey of Efficient Multimodal Learning}},
  year = {2026},
  url = {https://arxiv.org/abs/2609.19445},
  note = {FM038: TMLR 2026（arXiv journal reference）},
  urldate = {2026-10-03},
  eprint = {2609.19445},
  archivePrefix = {arXiv}
}

@misc{FM039,
  author = {Milad Abdollahzadeh and others},
  title = {{Revisit Multimodal Meta-Learning through the Lens of Multi-Task Learning}},
  year = {2021},
  url = {https://arxiv.org/abs/2110.14202},
  note = {FM039: NeurIPS 2021},
  urldate = {2026-10-03},
  eprint = {2110.14202},
  archivePrefix = {arXiv}
}

@misc{FM040,
  author = {Paul Christiano and others},
  title = {{Deep reinforcement learning from human preferences}},
  year = {2017},
  url = {https://arxiv.org/abs/1706.03741},
  note = {FM040: arXiv preprint（会議版未再照合）},
  urldate = {2026-10-03},
  eprint = {1706.03741},
  archivePrefix = {arXiv}
}

@misc{FM041,
  author = {Daniel M. Ziegler and others},
  title = {{Fine-Tuning Language Models from Human Preferences}},
  year = {2019},
  url = {https://arxiv.org/abs/1909.08593},
  note = {FM041: arXiv preprint},
  urldate = {2026-10-03},
  eprint = {1909.08593},
  archivePrefix = {arXiv}
}

@misc{FM042,
  author = {Nisan Stiennon and others},
  title = {{Learning to summarize from human feedback}},
  year = {2020},
  url = {https://arxiv.org/abs/2009.01325},
  note = {FM042: NeurIPS 2020},
  urldate = {2026-10-03},
  eprint = {2009.01325},
  archivePrefix = {arXiv}
}

@misc{FM043,
  author = {Long Ouyang and others},
  title = {{Training language models to follow instructions with human feedback}},
  year = {2022},
  url = {https://arxiv.org/abs/2203.02155},
  note = {FM043: arXiv preprint（会議版未再照合）},
  urldate = {2026-10-03},
  eprint = {2203.02155},
  archivePrefix = {arXiv}
}

@misc{FM044,
  author = {John Schulman and others},
  title = {{Proximal Policy Optimization Algorithms}},
  year = {2017},
  url = {https://arxiv.org/abs/1707.06347},
  note = {FM044: arXiv preprint},
  urldate = {2026-10-03},
  eprint = {1707.06347},
  archivePrefix = {arXiv}
}

@misc{FM045,
  author = {Yuntao Bai and others},
  title = {{Constitutional AI: Harmlessness from AI Feedback}},
  year = {2022},
  url = {https://arxiv.org/abs/2212.08073},
  note = {FM045: arXiv preprint},
  urldate = {2026-10-03},
  eprint = {2212.08073},
  archivePrefix = {arXiv}
}

@misc{FM046,
  author = {Rafael Rafailov and others},
  title = {{Direct Preference Optimization: Your Language Model is Secretly a Reward Model}},
  year = {2023},
  url = {https://arxiv.org/abs/2305.18290},
  note = {FM046: arXiv preprint（会議版未再照合）},
  urldate = {2026-10-03},
  eprint = {2305.18290},
  archivePrefix = {arXiv}
}

@misc{FM047,
  author = {Mohammad Gheshlaghi Azar and others},
  title = {{A General Theoretical Paradigm to Understand Learning from Human Preferences}},
  year = {2023},
  url = {https://arxiv.org/abs/2310.12036},
  note = {FM047: arXiv preprint（会議版未再照合）},
  urldate = {2026-10-03},
  eprint = {2310.12036},
  archivePrefix = {arXiv}
}

@misc{FM048,
  author = {Kawin Ethayarajh and others},
  title = {{KTO: Model Alignment as Prospect Theoretic Optimization}},
  year = {2024},
  url = {https://arxiv.org/abs/2402.01306},
  note = {FM048: ICML 2024},
  urldate = {2026-10-03},
  eprint = {2402.01306},
  archivePrefix = {arXiv}
}

@misc{FM049,
  author = {Yu Meng and others},
  title = {{SimPO: Simple Preference Optimization with a Reference-Free Reward}},
  year = {2024},
  url = {https://arxiv.org/abs/2405.14734},
  note = {FM049: NeurIPS 2024},
  urldate = {2026-10-03},
  eprint = {2405.14734},
  archivePrefix = {arXiv}
}

@misc{FM050,
  author = {Leo Gao and others},
  title = {{Scaling Laws for Reward Model Overoptimization}},
  year = {2022},
  url = {https://arxiv.org/abs/2210.10760},
  note = {FM050: arXiv preprint（会議版未再照合）},
  urldate = {2026-10-03},
  eprint = {2210.10760},
  archivePrefix = {arXiv}
}

@misc{FM051,
  author = {Stephen Casper and others},
  title = {{Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback}},
  year = {2023},
  url = {https://arxiv.org/abs/2307.15217},
  note = {FM051: arXiv preprint},
  urldate = {2026-10-03},
  eprint = {2307.15217},
  archivePrefix = {arXiv}
}

@misc{FM052,
  author = {Luca Viano and others},
  title = {{Direct Preference Optimization with Rating Information: Practical Algorithms and Provable Gains}},
  year = {2026},
  url = {https://arxiv.org/abs/2602.00603},
  note = {FM052: arXiv preprint},
  urldate = {2026-10-03},
  eprint = {2602.00603},
  archivePrefix = {arXiv}
}

@misc{FM053,
  author = {Pei-Chi Pan and others},
  title = {{Reward Modeling for Reinforcement Learning-Based LLM Reasoning: Design, Challenges, and Evaluation}},
  year = {2026},
  url = {https://arxiv.org/abs/2602.09305},
  note = {FM053: TMLR 2026（arXiv採択記載）},
  urldate = {2026-10-03},
  eprint = {2602.09305},
  archivePrefix = {arXiv}
}

@misc{FM054,
  author = {Jason Wei and others},
  title = {{Chain-of-Thought Prompting Elicits Reasoning in Large Language Models}},
  year = {2022},
  url = {https://arxiv.org/abs/2201.11903},
  note = {FM054: arXiv preprint（会議版未再照合）},
  urldate = {2026-10-03},
  eprint = {2201.11903},
  archivePrefix = {arXiv}
}

@misc{FM055,
  author = {Xuezhi Wang and others},
  title = {{Self-Consistency Improves Chain of Thought Reasoning in Language Models}},
  year = {2022},
  url = {https://arxiv.org/abs/2203.11171},
  note = {FM055: ICLR 2023},
  urldate = {2026-10-03},
  eprint = {2203.11171},
  archivePrefix = {arXiv}
}

@misc{FM056,
  author = {Shunyu Yao and others},
  title = {{Tree of Thoughts: Deliberate Problem Solving with Large Language Models}},
  year = {2023},
  url = {https://arxiv.org/abs/2305.10601},
  note = {FM056: NeurIPS 2023},
  urldate = {2026-10-03},
  eprint = {2305.10601},
  archivePrefix = {arXiv}
}

@misc{FM057,
  author = {Shunyu Yao and others},
  title = {{ReAct: Synergizing Reasoning and Acting in Language Models}},
  year = {2022},
  url = {https://arxiv.org/abs/2210.03629},
  note = {FM057: arXiv preprint（会議版未再照合）},
  urldate = {2026-10-03},
  eprint = {2210.03629},
  archivePrefix = {arXiv}
}

@misc{FM058,
  author = {Hunter Lightman and others},
  title = {{Let's Verify Step by Step}},
  year = {2023},
  url = {https://arxiv.org/abs/2305.20050},
  note = {FM058: arXiv preprint（会議版未再照合）},
  urldate = {2026-10-03},
  eprint = {2305.20050},
  archivePrefix = {arXiv}
}

@misc{FM059,
  author = {Charlie Snell and others},
  title = {{Scaling LLM Test-Time Compute Optimally can be More Effective than Scaling Model Parameters}},
  year = {2024},
  url = {https://arxiv.org/abs/2408.03314},
  note = {FM059: arXiv preprint（会議版未再照合）},
  urldate = {2026-10-03},
  eprint = {2408.03314},
  archivePrefix = {arXiv}
}

@misc{FM060,
  author = {Zhihong Shao and others},
  title = {{DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models}},
  year = {2024},
  url = {https://arxiv.org/abs/2402.03300},
  note = {FM060: arXiv preprint},
  urldate = {2026-10-03},
  eprint = {2402.03300},
  archivePrefix = {arXiv}
}

@misc{FM061,
  author = {DeepSeek-AI and others},
  title = {{DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning}},
  year = {2025},
  url = {https://arxiv.org/abs/2501.12948},
  note = {FM061: Nature 645, 633–638 (2025); 技術報告v1を本文確認},
  urldate = {2026-10-03},
  eprint = {2501.12948},
  archivePrefix = {arXiv}
}

@misc{FM062,
  author = {Niklas Muennighoff and others},
  title = {{s1: Simple test-time scaling}},
  year = {2025},
  url = {https://arxiv.org/abs/2501.19393},
  note = {FM062: EMNLP 2025},
  urldate = {2026-10-03},
  eprint = {2501.19393},
  archivePrefix = {arXiv}
}

@misc{FM063,
  author = {Jonas Geiping and others},
  title = {{Scaling up Test-Time Compute with Latent Reasoning: A Recurrent Depth Approach}},
  year = {2025},
  url = {https://arxiv.org/abs/2502.05171},
  note = {FM063: arXiv preprint},
  urldate = {2026-10-03},
  eprint = {2502.05171},
  archivePrefix = {arXiv}
}

@misc{FM064,
  author = {Qiying Yu and others},
  title = {{DAPO: An Open-Source LLM Reinforcement Learning System at Scale}},
  year = {2025},
  url = {https://arxiv.org/abs/2503.14476},
  note = {FM064: arXiv preprint},
  urldate = {2026-10-03},
  eprint = {2503.14476},
  archivePrefix = {arXiv}
}

@misc{FM065,
  author = {Yang Yue and others},
  title = {{Does Reinforcement Learning Really Incentivize Reasoning Capacity in LLMs Beyond the Base Model?}},
  year = {2025},
  url = {https://arxiv.org/abs/2504.13837},
  note = {FM065: NeurIPS 2025},
  urldate = {2026-10-03},
  eprint = {2504.13837},
  archivePrefix = {arXiv}
}

@misc{FM066,
  author = {Parshin Shojaee and others},
  title = {{The Illusion of Thinking: Understanding the Strengths and Limitations of Reasoning Models via the Lens of Problem Complexity}},
  year = {2025},
  url = {https://arxiv.org/abs/2506.06941},
  note = {FM066: NeurIPS 2025},
  urldate = {2026-10-03},
  eprint = {2506.06941},
  archivePrefix = {arXiv}
}

@misc{FM067,
  author = {Ismail Labiad and others},
  title = {{Beyond Repeated Sampling: Learning Search Policies for LLM Reasoning}},
  year = {2026},
  url = {https://arxiv.org/abs/2609.26704},
  note = {FM067: arXiv preprint},
  urldate = {2026-10-03},
  eprint = {2609.26704},
  archivePrefix = {arXiv}
}

@misc{FM068,
  author = {Qili Zhang and others},
  title = {{Learning to Prove, Not Just to Answer: Reinforcement Learning from Formal Verification for Natural-Language Logical Reasoning}},
  year = {2026},
  url = {https://arxiv.org/abs/2609.37203},
  note = {FM068: arXiv preprint},
  urldate = {2026-10-03},
  eprint = {2609.37203},
  archivePrefix = {arXiv}
}

@misc{FM069,
  author = {Ian J. Goodfellow and others},
  title = {{Generative Adversarial Networks}},
  year = {2014},
  url = {https://arxiv.org/abs/1406.2661},
  note = {FM069: arXiv preprint（会議題名はGenerative Adversarial Nets） / MD76: NeurIPS 2014, published;刊行題名はGenerative Adversarial Nets、記載題名はarXiv版},
  urldate = {2026-10-03},
  eprint = {1406.2661},
  archivePrefix = {arXiv}
}

@misc{FM070,
  author = {Alec Radford and others},
  title = {{Unsupervised Representation Learning with Deep Convolutional Generative Adversarial Networks}},
  year = {2015},
  url = {https://arxiv.org/abs/1511.06434},
  note = {FM070: arXiv preprint（会議版未再照合）},
  urldate = {2026-10-03},
  eprint = {1511.06434},
  archivePrefix = {arXiv}
}

@misc{FM071,
  author = {Martin Arjovsky and others},
  title = {{Wasserstein GAN}},
  year = {2017},
  url = {https://arxiv.org/abs/1701.07875},
  note = {FM071: arXiv preprint（会議版未再照合） / MD77: ICML 2017, published;刊行題名はWasserstein Generative Adversarial Networks、記載題名はarXiv版},
  urldate = {2026-10-03},
  eprint = {1701.07875},
  archivePrefix = {arXiv}
}

@misc{FM072,
  author = {Ishaan Gulrajani and others},
  title = {{Improved Training of Wasserstein GANs}},
  year = {2017},
  url = {https://arxiv.org/abs/1704.00028},
  note = {FM072: NeurIPS 2017 / MD78: NeurIPS 2017, published},
  urldate = {2026-10-03},
  eprint = {1704.00028},
  archivePrefix = {arXiv}
}

@misc{FM073,
  author = {Takeru Miyato and others},
  title = {{Spectral Normalization for Generative Adversarial Networks}},
  year = {2018},
  url = {https://arxiv.org/abs/1802.05957},
  note = {FM073: ICLR 2018 / MD79: ICLR 2018, published},
  urldate = {2026-10-03},
  eprint = {1802.05957},
  archivePrefix = {arXiv}
}

@misc{FM074,
  author = {Lars Mescheder and others},
  title = {{The Numerics of GANs}},
  year = {2017},
  url = {https://arxiv.org/abs/1705.10461},
  note = {FM074: arXiv preprint（会議版未再照合）},
  urldate = {2026-10-03},
  eprint = {1705.10461},
  archivePrefix = {arXiv}
}

@misc{FM075,
  author = {Lars Mescheder and others},
  title = {{Which Training Methods for GANs do actually Converge?}},
  year = {2018},
  url = {https://arxiv.org/abs/1801.04406},
  note = {FM075: ICML 2018},
  urldate = {2026-10-03},
  eprint = {1801.04406},
  archivePrefix = {arXiv}
}

@misc{FM077,
  author = {Andrew Brock and others},
  title = {{Large Scale GAN Training for High Fidelity Natural Image Synthesis}},
  year = {2018},
  url = {https://arxiv.org/abs/1809.11096},
  note = {FM077: arXiv preprint（会議版未再照合）},
  urldate = {2026-10-03},
  eprint = {1809.11096},
  archivePrefix = {arXiv}
}

@misc{FM078,
  author = {Tero Karras and others},
  title = {{A Style-Based Generator Architecture for Generative Adversarial Networks}},
  year = {2018},
  url = {https://arxiv.org/abs/1812.04948},
  note = {FM078: CVPR 2019},
  urldate = {2026-10-03},
  eprint = {1812.04948},
  archivePrefix = {arXiv}
}

@misc{FM079,
  author = {Tero Karras and others},
  title = {{Analyzing and Improving the Image Quality of StyleGAN}},
  year = {2019},
  url = {https://arxiv.org/abs/1912.04958},
  note = {FM079: arXiv preprint（会議版未再照合）},
  urldate = {2026-10-03},
  eprint = {1912.04958},
  archivePrefix = {arXiv}
}

@misc{FM080,
  author = {Tero Karras and others},
  title = {{Alias-Free Generative Adversarial Networks}},
  year = {2021},
  url = {https://arxiv.org/abs/2106.12423},
  note = {FM080: arXiv preprint（会議版未再照合）},
  urldate = {2026-10-03},
  eprint = {2106.12423},
  archivePrefix = {arXiv}
}

@misc{FM081,
  author = {Minguk Kang and others},
  title = {{Scaling up GANs for Text-to-Image Synthesis}},
  year = {2023},
  url = {https://arxiv.org/abs/2303.05511},
  note = {FM081: CVPR 2023},
  urldate = {2026-10-03},
  eprint = {2303.05511},
  archivePrefix = {arXiv}
}

@misc{FM082,
  author = {Yiwen Huang and others},
  title = {{The GAN is dead; long live the GAN! A Modern GAN Baseline}},
  year = {2024},
  url = {https://arxiv.org/abs/2501.05441},
  note = {FM082: NeurIPS 2024; arXiv掲載は2025},
  urldate = {2026-10-03},
  eprint = {2501.05441},
  archivePrefix = {arXiv}
}

@misc{FM083,
  author = {Jonathan Ho and others},
  title = {{Denoising Diffusion Probabilistic Models}},
  year = {2020},
  url = {https://arxiv.org/abs/2006.11239},
  note = {FM083: arXiv preprint（会議版未再照合） / MD23: NeurIPS 2020, published},
  urldate = {2026-10-03},
  eprint = {2006.11239},
  archivePrefix = {arXiv}
}

@misc{FM084,
  author = {Tianwei Yin and others},
  title = {{Improved Distribution Matching Distillation for Fast Image Synthesis}},
  year = {2024},
  url = {https://arxiv.org/abs/2405.14867},
  note = {FM084: arXiv preprint（会議版未再照合）},
  urldate = {2026-10-03},
  eprint = {2405.14867},
  archivePrefix = {arXiv}
}

@misc{FM085,
  author = {Saumya Gupta and others},
  title = {{Image Synthesis Using Spintronic Deep Convolutional Generative Adversarial Network}},
  year = {2026},
  url = {https://arxiv.org/abs/2601.01441},
  note = {FM085: arXiv preprint},
  urldate = {2026-10-03},
  eprint = {2601.01441},
  archivePrefix = {arXiv}
}

@misc{FM086,
  author = {Mohammad Shoeybi and others},
  title = {{Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism}},
  year = {2019},
  url = {https://arxiv.org/abs/1909.08053},
  note = {FM086: arXiv preprint / SY003: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {1909.08053},
  archivePrefix = {arXiv}
}

@misc{FM087,
  author = {Samyam Rajbhandari and others},
  title = {{ZeRO: Memory Optimizations Toward Training Trillion Parameter Models}},
  year = {2019},
  url = {https://arxiv.org/abs/1910.02054},
  note = {FM087: arXiv preprint（会議版未再照合） / SY001: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {1910.02054},
  archivePrefix = {arXiv}
}

@misc{FM088,
  author = {Edward J. Hu and others},
  title = {{LoRA: Low-Rank Adaptation of Large Language Models}},
  year = {2021},
  url = {https://arxiv.org/abs/2106.09685},
  note = {FM088: arXiv preprint（会議版未再照合）},
  urldate = {2026-10-03},
  eprint = {2106.09685},
  archivePrefix = {arXiv}
}

@misc{FM089,
  author = {Tim Dettmers and others},
  title = {{QLoRA: Efficient Finetuning of Quantized LLMs}},
  year = {2023},
  url = {https://arxiv.org/abs/2305.14314},
  note = {FM089: arXiv preprint（会議版未再照合）},
  urldate = {2026-10-03},
  eprint = {2305.14314},
  archivePrefix = {arXiv}
}

@misc{FM090,
  author = {Elias Frantar and others},
  title = {{GPTQ: Accurate Post-Training Quantization for Generative Pre-trained Transformers}},
  year = {2022},
  url = {https://arxiv.org/abs/2210.17323},
  note = {FM090: ICLR 2023},
  urldate = {2026-10-03},
  eprint = {2210.17323},
  archivePrefix = {arXiv}
}

@misc{FM091,
  author = {Ji Lin and others},
  title = {{AWQ: Activation-aware Weight Quantization for LLM Compression and Acceleration}},
  year = {2023},
  url = {https://arxiv.org/abs/2306.00978},
  note = {FM091: MLSys 2024},
  urldate = {2026-10-03},
  eprint = {2306.00978},
  archivePrefix = {arXiv}
}

@misc{FM092,
  author = {Woosuk Kwon and others},
  title = {{Efficient Memory Management for Large Language Model Serving with PagedAttention}},
  year = {2023},
  url = {https://arxiv.org/abs/2309.06180},
  note = {FM092: arXiv preprint（会議版未再照合）},
  urldate = {2026-10-03},
  eprint = {2309.06180},
  archivePrefix = {arXiv}
}

@misc{FM093,
  author = {Yaniv Leviathan and others},
  title = {{Fast Inference from Transformers via Speculative Decoding}},
  year = {2022},
  url = {https://arxiv.org/abs/2211.17192},
  note = {FM093: ICML 2023},
  urldate = {2026-10-03},
  eprint = {2211.17192},
  archivePrefix = {arXiv}
}

@misc{FM094,
  author = {Lianmin Zheng and others},
  title = {{SGLang: Efficient Execution of Structured Language Model Programs}},
  year = {2023},
  url = {https://arxiv.org/abs/2312.07104},
  note = {FM094: arXiv preprint（会議版未再照合）},
  urldate = {2026-10-03},
  eprint = {2312.07104},
  archivePrefix = {arXiv}
}

@misc{FM095,
  author = {Jiquan Ngiam and others},
  title = {{Multimodal Deep Learning}},
  year = {2011},
  url = {https://icml.cc/2011/papers/399_icmlpaper.pdf},
  note = {FM095: ICML 2011},
  urldate = {2026-10-03}
}

@misc{FM096,
  author = {Sören Richard Stahlschmidt and others},
  title = {{Multimodal deep learning for biomedical data fusion: a review}},
  year = {2022},
  url = {https://pmc.ncbi.nlm.nih.gov/articles/PMC8921642/},
  note = {FM096: Briefings in Bioinformatics 23(2), bbab569 (2022)},
  urldate = {2026-10-03}
}

@misc{FM097,
  author = {Jabeen Summaira and others},
  title = {{Recent Advances and Trends in Multimodal Deep Learning: A Review}},
  year = {2021},
  url = {https://arxiv.org/abs/2105.11087},
  note = {FM097: arXiv preprint},
  urldate = {2026-10-03},
  eprint = {2105.11087},
  archivePrefix = {arXiv}
}

@misc{FM098,
  author = {Tuomas Kynkäänniemi and others},
  title = {{Improved Precision and Recall Metric for Assessing Generative Models}},
  year = {2019},
  url = {https://arxiv.org/abs/1904.06991},
  note = {FM098: NeurIPS 2019},
  urldate = {2026-10-03},
  eprint = {1904.06991},
  archivePrefix = {arXiv}
}

@misc{FM099,
  author = {Gaurav Parmar and others},
  title = {{On Aliased Resizing and Surprising Subtleties in GAN Evaluation}},
  year = {2021},
  url = {https://arxiv.org/abs/2104.11222},
  note = {FM099: CVPR 2022},
  urldate = {2026-10-03},
  eprint = {2104.11222},
  archivePrefix = {arXiv}
}

@misc{FM100,
  author = {Nicholas Carlini and others},
  title = {{Extracting Training Data from Large Language Models}},
  year = {2020},
  url = {https://arxiv.org/abs/2012.07805},
  note = {FM100: arXiv preprint（会議版未再照合）},
  urldate = {2026-10-03},
  eprint = {2012.07805},
  archivePrefix = {arXiv}
}

@misc{FM101,
  author = {Han Zhang and others},
  title = {{StackGAN: Text to Photo-realistic Image Synthesis with Stacked Generative Adversarial Networks}},
  year = {2016},
  url = {https://arxiv.org/abs/1612.03242},
  note = {FM101: ICCV 2017},
  urldate = {2026-10-03},
  eprint = {1612.03242},
  archivePrefix = {arXiv}
}

@misc{FM102,
  author = {Atsuhiro Noguchi and others},
  title = {{RGBD-GAN: Unsupervised 3D Representation Learning From Natural Image Datasets via RGBD Image Synthesis}},
  year = {2019},
  url = {https://arxiv.org/abs/1909.12573},
  note = {FM102: ICLR 2020},
  urldate = {2026-10-03},
  eprint = {1909.12573},
  archivePrefix = {arXiv}
}

@misc{FM103,
  author = {Cyprien de Masson d'Autume and others},
  title = {{Training language GANs from Scratch}},
  year = {2019},
  url = {https://arxiv.org/abs/1905.09922},
  note = {FM103: arXiv preprint（会議版未再照合）},
  urldate = {2026-10-03},
  eprint = {1905.09922},
  archivePrefix = {arXiv}
}

@misc{FM104,
  author = {Vineet Kosaraju and others},
  title = {{Social-BiGAT: Multimodal Trajectory Forecasting using Bicycle-GAN and Graph Attention Networks}},
  year = {2019},
  url = {https://arxiv.org/abs/1907.03395},
  note = {FM104: arXiv preprint（会議版未再照合）},
  urldate = {2026-10-03},
  eprint = {1907.03395},
  archivePrefix = {arXiv}
}

@misc{FM105,
  author = {Shanchuan Lin and others},
  title = {{Diffusion Adversarial Post-Training for One-Step Video Generation}},
  year = {2025},
  url = {https://arxiv.org/abs/2501.08316},
  note = {FM105: ICML 2025},
  urldate = {2026-10-03},
  eprint = {2501.08316},
  archivePrefix = {arXiv}
}

@misc{FM106,
  author = {William Nixon and others},
  title = {{A Year in LLM Serving: Workload Evolution, Caching and Load-Balancing}},
  year = {2026},
  url = {https://arxiv.org/abs/2608.13573},
  note = {FM106: arXiv preprint},
  urldate = {2026-10-03},
  eprint = {2608.13573},
  archivePrefix = {arXiv}
}

@misc{FM107,
  author = {Ian Goodfellow},
  title = {{NIPS 2016 Tutorial: Generative Adversarial Networks}},
  year = {2016},
  url = {https://arxiv.org/abs/1701.00160},
  note = {FM107: NIPS 2016 tutorial report; arXiv識別子は1701、Submittedは2016-12-31},
  urldate = {2026-10-03},
  eprint = {1701.00160},
  archivePrefix = {arXiv}
}

@misc{FM108,
  author = {Bisakha Ray and others},
  title = {{Information content and analysis methods for Multi-Modal High-Throughput Biomedical Data}},
  year = {2014},
  url = {https://www.nature.com/articles/srep04411},
  note = {FM108: Scientific Reports 4, 4411 (2014)},
  urldate = {2026-10-03}
}

@misc{FM109,
  author = {Janani Venugopalan and others},
  title = {{Multimodal deep learning models for early detection of Alzheimer’s disease stage}},
  year = {2021},
  url = {https://pmc.ncbi.nlm.nih.gov/articles/PMC7864942/},
  note = {FM109: Scientific Reports 11, 3254 (2021)},
  urldate = {2026-10-03}
}

@misc{FM110,
  author = {福水健次},
  title = {{深層生成モデルによる統計的推論}},
  year = {2019},
  url = {https://www.ism.ac.jp/openhouse/2019/index/ISM-75-tutorial_fukumizu_open.pdf},
  note = {FM110: 統計数理研究所創立75周年記念チュートリアル講演、2019-06-05},
  urldate = {2026-10-03}
}

@misc{FM111,
  author = {Microsoft DeepSpeed Team},
  title = {{DeepSpeed: 深層学習の訓練と推論を劇的に高速化するフレームワーク}},
  year = {2023},
  url = {https://www.deepspeed.ai/assets/files/DeepSpeed_Overview_Japanese_2023Jun7th.pdf},
  note = {FM111: 公式日本語概要スライド、2023-06-07},
  urldate = {2026-10-03}
}

@misc{FM112,
  author = {Denny Zhou},
  title = {{LLM Reasoning: Key Ideas and Limitations}},
  year = {2024},
  url = {https://dennyzhou.github.io/LLM-Reasoning-Berkeley.pdf},
  note = {FM112: 著者公開講義スライド（表紙のBerkeley講演年2024を採用）},
  urldate = {2026-10-03}
}

@misc{MD01,
  author = {Atilim Gunes Baydin and Barak A. Pearlmutter and Alexey Andreyevich Radul and Jeffrey Mark Siskind},
  title = {{Automatic Differentiation in Machine Learning: a Survey}},
  year = {2018},
  url = {https://www.jmlr.org/papers/v18/17-468.html},
  note = {MD01: JMLR 18(153), published},
  urldate = {2026-10-03}
}

@misc{MD02,
  author = {Barak A. Pearlmutter},
  title = {{Fast Exact Multiplication by the Hessian}},
  year = {1994},
  url = {https://doi.org/10.1162/neco.1994.6.1.147},
  note = {MD02: Neural Computation 6(1), published},
  urldate = {2026-10-03}
}

@misc{MD03,
  author = {M. F. Hutchinson},
  title = {{A Stochastic Estimator of the Trace of the Influence Matrix for Laplacian Smoothing Splines}},
  year = {1989},
  url = {https://doi.org/10.1080/03610918908812806},
  note = {MD03: Communications in Statistics - Simulation and Computation 18(3), published},
  urldate = {2026-10-03}
}

@misc{MD04,
  author = {Raphael A. Meyer and Cameron Musco and Christopher Musco and David P. Woodruff},
  title = {{Hutch++: Optimal Stochastic Trace Estimation}},
  year = {2021},
  url = {https://arxiv.org/abs/2010.09649},
  note = {MD04: SOSA 2021, published (preprint 2020)},
  urldate = {2026-10-03},
  eprint = {2010.09649},
  archivePrefix = {arXiv}
}

@misc{MD05,
  author = {Felix Dangel and Frederik Kunstner and Philipp Hennig},
  title = {{BackPACK: Packing more into Backprop}},
  year = {2020},
  url = {https://arxiv.org/abs/1912.10985},
  note = {MD05: ICLR 2020, published (preprint 2019)},
  urldate = {2026-10-03},
  eprint = {1912.10985},
  archivePrefix = {arXiv}
}

@misc{MD06,
  author = {Behrooz Ghorbani and Shankar Krishnan and Ying Xiao},
  title = {{An Investigation into Neural Net Optimization via Hessian Eigenvalue Density}},
  year = {2019},
  url = {https://arxiv.org/abs/1901.10159},
  note = {MD06: ICML 2019, published},
  urldate = {2026-10-03},
  eprint = {1901.10159},
  archivePrefix = {arXiv}
}

@misc{MD07,
  author = {Frederik Kunstner and Lukas Balles and Philipp Hennig},
  title = {{Limitations of the Empirical Fisher Approximation for Natural Gradient Descent}},
  year = {2019},
  url = {https://arxiv.org/abs/1905.12558},
  note = {MD07: NeurIPS 2019, published},
  urldate = {2026-10-03},
  eprint = {1905.12558},
  archivePrefix = {arXiv}
}

@misc{MD09,
  author = {Guillaume Dalle and Adrian Hill},
  title = {{A Common Interface for Automatic Differentiation}},
  year = {2026},
  url = {https://www.jmlr.org/papers/v27/25-1024.html},
  note = {MD09: JMLR 27(25), published January 2026},
  urldate = {2026-10-03}
}

@misc{MD10,
  author = {Arthur Jacot and Franck Gabriel and Clément Hongler},
  title = {{Neural Tangent Kernel: Convergence and Generalization in Neural Networks}},
  year = {2018},
  url = {https://arxiv.org/abs/1806.07572},
  note = {MD10: NeurIPS 2018, published},
  urldate = {2026-10-03},
  eprint = {1806.07572},
  archivePrefix = {arXiv}
}

@misc{MD11,
  author = {Jaehoon Lee and Lechao Xiao and Samuel S. Schoenholz and Yasaman Bahri and Roman Novak and Jascha Sohl-Dickstein and Jeffrey Pennington},
  title = {{Wide Neural Networks of Any Depth Evolve as Linear Models Under Gradient Descent}},
  year = {2019},
  url = {https://arxiv.org/abs/1902.06720},
  note = {MD11: NeurIPS 2019, published},
  urldate = {2026-10-03},
  eprint = {1902.06720},
  archivePrefix = {arXiv}
}

@misc{MD12,
  author = {Lenaic Chizat and Edouard Oyallon and Francis Bach},
  title = {{On Lazy Training in Differentiable Programming}},
  year = {2019},
  url = {https://arxiv.org/abs/1812.07956},
  note = {MD12: NeurIPS 2019, published (preprint 2018)},
  urldate = {2026-10-03},
  eprint = {1812.07956},
  archivePrefix = {arXiv}
}

@misc{MD13,
  author = {Greg Yang and Edward J. Hu},
  title = {{Feature Learning in Infinite-Width Neural Networks}},
  year = {2021},
  url = {https://arxiv.org/abs/2011.14522},
  note = {MD13: ICML 2021, published (preprint 2020)},
  urldate = {2026-10-03},
  eprint = {2011.14522},
  archivePrefix = {arXiv}
}

@misc{MD14,
  author = {Greg Yang and Edward J. Hu and Igor Babuschkin and Szymon Sidor and Xiaodong Liu and David Farhi and Nick Ryder and Jakub Pachocki and Weizhu Chen and Jianfeng Gao},
  title = {{Tensor Programs V: Tuning Large Neural Networks via Zero-Shot Hyperparameter Transfer}},
  year = {2021},
  url = {https://arxiv.org/abs/2203.03466},
  note = {MD14: NeurIPS 2021, published; arXiv登録2022（刊行年と登録年が異なる）},
  urldate = {2026-10-03},
  eprint = {2203.03466},
  archivePrefix = {arXiv}
}

@misc{MD15,
  author = {Mikhail Belkin and Daniel Hsu and Siyuan Ma and Soumik Mandal},
  title = {{Reconciling modern machine learning practice and the bias-variance trade-off}},
  year = {2019},
  url = {https://arxiv.org/abs/1812.11118},
  note = {MD15: PNAS 2019, published (preprint 2018)},
  urldate = {2026-10-03},
  eprint = {1812.11118},
  archivePrefix = {arXiv}
}

@misc{MD16,
  author = {Preetum Nakkiran and Gal Kaplun and Yamini Bansal and Tristan Yang and Boaz Barak and Ilya Sutskever},
  title = {{Deep Double Descent: Where Bigger Models and More Data Hurt}},
  year = {2020},
  url = {https://arxiv.org/abs/1912.02292},
  note = {MD16: ICLR 2020, published; arXiv初稿2019 / MD29: ICLR 2023, published; arXiv初稿2022},
  urldate = {2026-10-03},
  eprint = {2209.03003},
  archivePrefix = {arXiv}
}

@misc{MD17,
  author = {Trevor Hastie and Andrea Montanari and Saharon Rosset and Ryan J. Tibshirani},
  title = {{Surprises in High-Dimensional Ridgeless Least Squares Interpolation}},
  year = {2022},
  url = {https://arxiv.org/abs/1903.08560},
  note = {MD17: Annals of Statistics 50(2), published (preprint 2019)},
  urldate = {2026-10-03},
  eprint = {1903.08560},
  archivePrefix = {arXiv}
}

@misc{MD18,
  author = {Peter L. Bartlett and Philip M. Long and Gábor Lugosi and Alexander Tsigler},
  title = {{Benign Overfitting in Linear Regression}},
  year = {2020},
  url = {https://arxiv.org/abs/1906.11300},
  note = {MD18: PNAS 2020, published (preprint 2019)},
  urldate = {2026-10-03},
  eprint = {1906.11300},
  archivePrefix = {arXiv}
}

@misc{MD19,
  author = {Jonathan Plenk and Sergio Calvo-Ordonez and Alvaro Cartea and Yarin Gal and Mark van der Wilk and Kamil Ciosek},
  title = {{The Neural Tangent Kernel for Classification}},
  year = {2026},
  url = {https://arxiv.org/abs/2605.17606},
  note = {MD19: preprint; submitted 2026-05-17, v2 2026-05-22},
  urldate = {2026-10-03},
  eprint = {2605.17606},
  archivePrefix = {arXiv}
}

@misc{MD20,
  author = {Zixiang Chen and Yuan Cao and Quanquan Gu and Tong Zhang},
  title = {{A Generalized Neural Tangent Kernel Analysis for Two-layer Neural Networks}},
  year = {2020},
  url = {https://arxiv.org/abs/2002.04026},
  note = {MD20: NeurIPS 2020, published},
  urldate = {2026-10-03},
  eprint = {2002.04026},
  archivePrefix = {arXiv}
}

@misc{MD21,
  author = {Kenji Kawaguchi and Qingyun Sun},
  title = {{A Recipe for Global Convergence Guarantee in Deep Neural Networks}},
  year = {2021},
  url = {https://arxiv.org/abs/2104.05785},
  note = {MD21: AAAI 2021, published},
  urldate = {2026-10-03},
  eprint = {2104.05785},
  archivePrefix = {arXiv}
}

@misc{MD22,
  author = {Shun-ichi Amari},
  title = {{Any Target Function Exists in a Neighborhood of Any Sufficiently Wide Random Network: A Geometrical Perspective}},
  year = {2020},
  url = {https://arxiv.org/abs/2001.06931},
  note = {MD22: Neural Computation 32(8), published},
  urldate = {2026-10-03},
  eprint = {2001.06931},
  archivePrefix = {arXiv}
}

@misc{MD24,
  author = {Jascha Sohl-Dickstein and Eric A. Weiss and Niru Maheswaranathan and Surya Ganguli},
  title = {{Deep Unsupervised Learning using Nonequilibrium Thermodynamics}},
  year = {2015},
  url = {https://arxiv.org/abs/1503.03585},
  note = {MD24: ICML 2015, published},
  urldate = {2026-10-03},
  eprint = {1503.03585},
  archivePrefix = {arXiv}
}

@misc{MD25,
  author = {Yang Song and Jascha Sohl-Dickstein and Diederik P. Kingma and Abhishek Kumar and Stefano Ermon and Ben Poole},
  title = {{Score-Based Generative Modeling through Stochastic Differential Equations}},
  year = {2021},
  url = {https://arxiv.org/abs/2011.13456},
  note = {MD25: ICLR 2021, published (preprint 2020)},
  urldate = {2026-10-03},
  eprint = {2011.13456},
  archivePrefix = {arXiv}
}

@misc{MD26,
  author = {Ricky T. Q. Chen and Yulia Rubanova and Jesse Bettencourt and David Duvenaud},
  title = {{Neural Ordinary Differential Equations}},
  year = {2018},
  url = {https://arxiv.org/abs/1806.07366},
  note = {MD26: NeurIPS 2018, published},
  urldate = {2026-10-03},
  eprint = {1806.07366},
  archivePrefix = {arXiv}
}

@misc{MD27,
  author = {Will Grathwohl and Ricky T. Q. Chen and Jesse Bettencourt and Ilya Sutskever and David Duvenaud},
  title = {{FFJORD: Free-form Continuous Dynamics for Scalable Reversible Generative Models}},
  year = {2019},
  url = {https://arxiv.org/abs/1810.01367},
  note = {MD27: ICLR 2019, published; arXiv初稿2018},
  urldate = {2026-10-03},
  eprint = {1810.01367},
  archivePrefix = {arXiv}
}

@misc{MD28,
  author = {Yaron Lipman and Ricky T. Q. Chen and Heli Ben-Hamu and Maximilian Nickel and Matt Le},
  title = {{Flow Matching for Generative Modeling}},
  year = {2023},
  url = {https://arxiv.org/abs/2210.02747},
  note = {MD28: ICLR 2023, published (preprint 2022)},
  urldate = {2026-10-03},
  eprint = {2210.02747},
  archivePrefix = {arXiv}
}

@misc{MD30,
  author = {Alexander Tong and Kilian Fatras and Nikolay Malkin and Guillaume Huguet and Yanlei Zhang and Jarrid Rector-Brooks and Guy Wolf and Yoshua Bengio},
  title = {{Improving and generalizing flow-based generative models with minibatch optimal transport}},
  year = {2024},
  url = {https://arxiv.org/abs/2302.00482},
  note = {MD30: TMLR 2024, published (preprint 2023)},
  urldate = {2026-10-03},
  eprint = {2302.00482},
  archivePrefix = {arXiv}
}

@misc{MD31,
  author = {Michael S. Albergo and Nicholas M. Boffi and Eric Vanden-Eijnden},
  title = {{Stochastic Interpolants: A Unifying Framework for Flows and Diffusions}},
  year = {2025},
  url = {https://arxiv.org/abs/2303.08797},
  note = {MD31: JMLR 26, published September 2025; arXiv初稿2023},
  urldate = {2026-10-03},
  eprint = {2303.08797},
  archivePrefix = {arXiv}
}

@misc{MD32,
  author = {Yang Song and Prafulla Dhariwal and Mark Chen and Ilya Sutskever},
  title = {{Consistency Models}},
  year = {2023},
  url = {https://arxiv.org/abs/2303.01469},
  note = {MD32: ICML 2023, published},
  urldate = {2026-10-03},
  eprint = {2303.01469},
  archivePrefix = {arXiv}
}

@misc{MD33,
  author = {Patrick Esser and Sumith Kulal and Andreas Blattmann and Rahim Entezari and Jonas Müller and Harry Saini and Yam Levi and Dominik Lorenz and Axel Sauer and Frederic Boesel and Dustin Podell and Tim Dockhorn and Zion English and Kyle Lacey and Alex Goodwin and Yannik Marek and Robin Rombach},
  title = {{Scaling Rectified Flow Transformers for High-Resolution Image Synthesis}},
  year = {2024},
  url = {https://arxiv.org/abs/2403.03206},
  note = {MD33: ICML 2024, published;ここではarXiv版の著者一覧を記載},
  urldate = {2026-10-03},
  eprint = {2403.03206},
  archivePrefix = {arXiv}
}

@misc{MD34,
  author = {Itai Gat and Tal Remez and Neta Shaul and Felix Kreuk and Ricky T. Q. Chen and Gabriel Synnaeve and Yossi Adi and Yaron Lipman},
  title = {{Discrete Flow Matching}},
  year = {2024},
  url = {https://arxiv.org/abs/2407.15595},
  note = {MD34: NeurIPS 2024, published},
  urldate = {2026-10-03},
  eprint = {2407.15595},
  archivePrefix = {arXiv}
}

@misc{MD35,
  author = {Yaron Lipman and Marton Havasi and Peter Holderrieth and Neta Shaul and Matt Le and Brian Karrer and Ricky T. Q. Chen and David Lopez-Paz and Heli Ben-Hamu and Itai Gat},
  title = {{Flow Matching Guide and Code}},
  year = {2024},
  url = {https://arxiv.org/abs/2412.06264},
  note = {MD35: arXiv guide 2024},
  urldate = {2026-10-03},
  eprint = {2412.06264},
  archivePrefix = {arXiv}
}

@misc{MD36,
  author = {Zhengyang Geng and Mingyang Deng and Xingjian Bai and J. Zico Kolter and Kaiming He},
  title = {{Mean Flows for One-step Generative Modeling}},
  year = {2025},
  url = {https://arxiv.org/abs/2505.13447},
  note = {MD36: NeurIPS 2025, proceedings PDF確認},
  urldate = {2026-10-03},
  eprint = {2505.13447},
  archivePrefix = {arXiv}
}

@misc{MD37,
  author = {Zhengyang Geng and Yiyang Lu and Zongze Wu and Eli Shechtman and J. Zico Kolter and Kaiming He},
  title = {{Improved Mean Flows: On the Challenges of Fastforward Generative Models}},
  year = {2025},
  url = {https://arxiv.org/abs/2512.02012},
  note = {MD37: technical report; submitted 2025-12-01, v2 2026-05-09},
  urldate = {2026-10-03},
  eprint = {2512.02012},
  archivePrefix = {arXiv}
}

@misc{MD38,
  author = {Mudit Gaur and Prashant Trivedi and Shuchin Aeron and Amrit Singh Bedi and George K. Atia and Vaneet Aggarwal},
  title = {{Generative Modeling with Continuous Flows: Sample Complexity of Flow Matching}},
  year = {2025},
  url = {https://arxiv.org/abs/2512.01286},
  note = {MD38: preprint; submitted 2025-12-01},
  urldate = {2026-10-03},
  eprint = {2512.01286},
  archivePrefix = {arXiv}
}

@misc{MD39,
  author = {Gabriel Peyré and Marco Cuturi},
  title = {{Computational Optimal Transport}},
  year = {2019},
  url = {https://arxiv.org/abs/1803.00567},
  note = {MD39: Foundations and Trends in Machine Learning 11(5-6), published},
  urldate = {2026-10-03},
  eprint = {1803.00567},
  archivePrefix = {arXiv}
}

@misc{MD40,
  author = {Marco Cuturi},
  title = {{Sinkhorn Distances: Lightspeed Computation of Optimal Transportation Distances}},
  year = {2013},
  url = {https://arxiv.org/abs/1306.0895},
  note = {MD40: NeurIPS 2013, published;刊行版題名末尾はOptimal Transport、掲載題名はarXiv版},
  urldate = {2026-10-03},
  eprint = {1306.0895},
  archivePrefix = {arXiv}
}

@misc{MD41,
  author = {Jean Feydy and Thibault Séjourné and François-Xavier Vialard and Shun-ichi Amari and Alain Trouvé and Gabriel Peyré},
  title = {{Interpolating between Optimal Transport and MMD using Sinkhorn Divergences}},
  year = {2019},
  url = {https://proceedings.mlr.press/v89/feydy19a.html},
  note = {MD41: AISTATS 2019, published},
  urldate = {2026-10-03}
}

@misc{MD42,
  author = {Nicolas Courty and Rémi Flamary and Devis Tuia and Alain Rakotomamonjy},
  title = {{Optimal Transport for Domain Adaptation}},
  year = {2017},
  url = {https://arxiv.org/abs/1507.00504},
  note = {MD42: IEEE TPAMI 39(9):1853–1865, published; arXiv初稿2015},
  urldate = {2026-10-03},
  eprint = {1507.00504},
  archivePrefix = {arXiv}
}

@misc{MD43,
  author = {Nicolas Courty and Rémi Flamary and Amaury Habrard and Alain Rakotomamonjy},
  title = {{Joint Distribution Optimal Transportation for Domain Adaptation}},
  year = {2017},
  url = {https://arxiv.org/abs/1705.08848},
  note = {MD43: NeurIPS 2017, published},
  urldate = {2026-10-03},
  eprint = {1705.08848},
  archivePrefix = {arXiv}
}

@misc{MD44,
  author = {Bharath Bhushan Damodaran and Benjamin Kellenberger and Rémi Flamary and Devis Tuia and Nicolas Courty},
  title = {{DeepJDOT: Deep Joint Distribution Optimal Transport for Unsupervised Domain Adaptation}},
  year = {2018},
  url = {https://arxiv.org/abs/1803.10081},
  note = {MD44: ECCV 2018, published},
  urldate = {2026-10-03},
  eprint = {1803.10081},
  archivePrefix = {arXiv}
}

@misc{MD45,
  author = {Lenaic Chizat and Gabriel Peyré and Bernhard Schmitzer and François-Xavier Vialard},
  title = {{Scaling Algorithms for Unbalanced Transport Problems}},
  year = {2018},
  url = {https://arxiv.org/abs/1607.05816},
  note = {MD45: Mathematics of Computation 87(314):2563–2609, published; arXiv初稿2016、刊行版題名にはOptimalが加わる},
  urldate = {2026-10-03},
  eprint = {1607.05816},
  archivePrefix = {arXiv}
}

@misc{MD46,
  author = {Kilian Fatras and Thibault Séjourné and Nicolas Courty and Rémi Flamary},
  title = {{Unbalanced minibatch Optimal Transport; applications to Domain Adaptation}},
  year = {2021},
  url = {https://arxiv.org/abs/2103.03606},
  note = {MD46: ICML 2021, published;著者順は論文PDF準拠},
  urldate = {2026-10-03},
  eprint = {2103.03606},
  archivePrefix = {arXiv}
}

@misc{MD47,
  author = {Yogesh Balaji and Rama Chellappa and Soheil Feizi},
  title = {{Robust Optimal Transport with Applications in Generative Modeling and Domain Adaptation}},
  year = {2020},
  url = {https://arxiv.org/abs/2010.05862},
  note = {MD47: NeurIPS 2020, published},
  urldate = {2026-10-03},
  eprint = {2010.05862},
  archivePrefix = {arXiv}
}

@misc{MD48,
  author = {Sloan Nietert and Rachel Cummings and Ziv Goldfeld},
  title = {{Outlier-Robust Optimal Transport: Duality, Structure, and Statistical Analysis}},
  year = {2022},
  url = {https://arxiv.org/abs/2111.01361},
  note = {MD48: AISTATS 2022, published;著者順はarXiv準拠},
  urldate = {2026-10-03},
  eprint = {2111.01361},
  archivePrefix = {arXiv}
}

@misc{MD49,
  author = {Jin Zhang and Mingyang Zhao and Bing Liu and Xin Jiang},
  title = {{Sinkhorn-CPD: Robust point cloud registration via unbalanced entropic optimal transport}},
  year = {2026},
  url = {https://arxiv.org/abs/2606.16672},
  note = {MD49: arXiv公開 2026-06-15; Computer-Aided Design 199 (2026), 104104との書誌記載あり、雑誌刊行日未照合},
  urldate = {2026-10-03},
  eprint = {2606.16672},
  archivePrefix = {arXiv}
}

@misc{MD50,
  author = {Nicholas Metropolis and Arianna W. Rosenbluth and Marshall N. Rosenbluth and Augusta H. Teller and Edward Teller},
  title = {{Equation of State Calculations by Fast Computing Machines}},
  year = {1953},
  url = {https://doi.org/10.1063/1.1699114},
  note = {MD50: Journal of Chemical Physics 21(6), published},
  urldate = {2026-10-03}
}

@misc{MD51,
  author = {W. K. Hastings},
  title = {{Monte Carlo sampling methods using Markov chains and their applications}},
  year = {1970},
  url = {https://doi.org/10.1093/biomet/57.1.97},
  note = {MD51: Biometrika 57(1), published},
  urldate = {2026-10-03}
}

@misc{MD52,
  author = {Radford M. Neal},
  title = {{MCMC using Hamiltonian dynamics}},
  year = {2012},
  url = {https://arxiv.org/abs/1206.1901},
  note = {MD52: arXiv公開版2012 (Handbook章2011)},
  urldate = {2026-10-03},
  eprint = {1206.1901},
  archivePrefix = {arXiv}
}

@misc{MD53,
  author = {Matthew D. Hoffman and Andrew Gelman},
  title = {{The No-U-Turn Sampler: Adaptively Setting Path Lengths in Hamiltonian Monte Carlo}},
  year = {2014},
  url = {https://jmlr.org/papers/v15/hoffman14a.html},
  note = {MD53: JMLR 15(47), published},
  urldate = {2026-10-03}
}

@misc{MD54,
  author = {Michael Betancourt},
  title = {{A Conceptual Introduction to Hamiltonian Monte Carlo}},
  year = {2017},
  url = {https://arxiv.org/abs/1701.02434},
  note = {MD54: arXiv review 2017, revised 2018},
  urldate = {2026-10-03},
  eprint = {1701.02434},
  archivePrefix = {arXiv}
}

@misc{MD55,
  author = {Max Welling and Yee Whye Teh},
  title = {{Bayesian Learning via Stochastic Gradient Langevin Dynamics}},
  year = {2011},
  url = {https://www.maths.ox.ac.uk/node/24560},
  note = {MD55: ICML 2011, published;著者所属機関の原著書誌・abstract確認},
  urldate = {2026-10-03}
}

@misc{MD56,
  author = {Tianqi Chen and Emily Fox and Carlos Guestrin},
  title = {{Stochastic Gradient Hamiltonian Monte Carlo}},
  year = {2014},
  url = {https://proceedings.mlr.press/v32/cheni14.html},
  note = {MD56: ICML 2014, published},
  urldate = {2026-10-03}
}

@misc{MD57,
  author = {Aki Vehtari and Andrew Gelman and Daniel Simpson and Bob Carpenter and Paul-Christian Bürkner},
  title = {{Rank-normalization, folding, and localization: An improved R-hat for assessing convergence of MCMC}},
  year = {2021},
  url = {https://arxiv.org/abs/1903.08008},
  note = {MD57: Bayesian Analysis, published (preprint 2019)},
  urldate = {2026-10-03},
  eprint = {1903.08008},
  archivePrefix = {arXiv}
}

@misc{MD58,
  author = {Nawaf Bou-Rabee and Bob Carpenter and Tore Selland Kleppe and Sifan Liu},
  title = {{The Within-Orbit Adaptive Leapfrog No-U-Turn Sampler}},
  year = {2026},
  url = {https://jmlr.org/papers/v27/25-1452.html},
  note = {MD58: JMLR 27(113), published March 2026},
  urldate = {2026-10-03}
}

@misc{MD59,
  author = {P.-A. Absil and Robert Mahony and Rodolphe Sepulchre},
  title = {{Optimization Algorithms on Matrix Manifolds}},
  year = {2008},
  url = {https://assets.press.princeton.edu/chapters/absil/Absil_pp.%20i-vi.pdf},
  note = {MD59: Princeton University Press book, published},
  urldate = {2026-10-03}
}

@misc{MD60,
  author = {Nicolas Boumal},
  title = {{An Introduction to Optimization on Smooth Manifolds}},
  year = {2023},
  url = {https://www.nicolasboumal.net/book/},
  note = {MD60: Cambridge University Press book, published},
  urldate = {2026-10-03}
}

@misc{MD61,
  author = {Tim Large and Yang Liu and Minyoung Huh and Hyojin Bahng and Phillip Isola and Jeremy Bernstein},
  title = {{Scalable Optimization in the Modular Norm}},
  year = {2024},
  url = {https://arxiv.org/abs/2405.14813},
  note = {MD61: NeurIPS 2024, proceedings PDF確認},
  urldate = {2026-10-03},
  eprint = {2405.14813},
  archivePrefix = {arXiv}
}

@misc{MD62,
  author = {Jeremy Bernstein},
  title = {{Modular Manifolds}},
  year = {2025},
  url = {https://thinkingmachines.ai/blog/modular-manifolds/},
  note = {MD62: 著者公式研究記事 2025-09-26;査読論文ではない},
  urldate = {2026-10-03}
}

@misc{MD63,
  author = {Aapo Hyvärinen and Erkki Oja},
  title = {{Independent component analysis: algorithms and applications}},
  year = {2000},
  url = {https://www.sciencedirect.com/science/article/pii/S0893608000000265},
  note = {MD63: Neural Networks 13, published},
  urldate = {2026-10-03}
}

@misc{MD64,
  author = {Aapo Hyvärinen and Hiroaki Sasaki and Richard E. Turner},
  title = {{Nonlinear ICA Using Auxiliary Variables and Generalized Contrastive Learning}},
  year = {2019},
  url = {https://proceedings.mlr.press/v89/hyvarinen19a.html},
  note = {MD64: AISTATS 2019, published},
  urldate = {2026-10-03}
}

@misc{MD65,
  author = {Ilyes Khemakhem and Diederik P. Kingma and Ricardo Pio Monti and Aapo Hyvärinen},
  title = {{Variational Autoencoders and Nonlinear ICA: A Unifying Framework}},
  year = {2020},
  url = {https://arxiv.org/abs/1907.04809},
  note = {MD65: AISTATS 2020, published},
  urldate = {2026-10-03},
  eprint = {1907.04809},
  archivePrefix = {arXiv}
}

@misc{MD66,
  author = {Francesco Locatello and Stefan Bauer and Mario Lucic and Gunnar Rätsch and Sylvain Gelly and Bernhard Schölkopf and Olivier Bachem},
  title = {{Challenging Common Assumptions in the Unsupervised Learning of Disentangled Representations}},
  year = {2019},
  url = {https://arxiv.org/abs/1811.12359},
  note = {MD66: ICML 2019, published},
  urldate = {2026-10-03},
  eprint = {1811.12359},
  archivePrefix = {arXiv}
}

@misc{MD67,
  author = {Bernhard Schölkopf and Francesco Locatello and Stefan Bauer and Nan Rosemary Ke and Nal Kalchbrenner and Anirudh Goyal and Yoshua Bengio},
  title = {{Towards Causal Representation Learning}},
  year = {2021},
  url = {https://arxiv.org/abs/2102.11107},
  note = {MD67: Proceedings of the IEEE 109(5), published;雑誌題名はToward},
  urldate = {2026-10-03},
  eprint = {2102.11107},
  archivePrefix = {arXiv}
}

@misc{MD68,
  author = {Siyuan Guo and Viktor Tóth and Bernhard Schölkopf and Ferenc Huszár},
  title = {{Causal de Finetti: On the Identification of Invariant Causal Structure in Exchangeable Data}},
  year = {2023},
  url = {https://arxiv.org/abs/2203.15756},
  note = {MD68: NeurIPS 2023, published (preprint 2022)},
  urldate = {2026-10-03},
  eprint = {2203.15756},
  archivePrefix = {arXiv}
}

@misc{MD69,
  author = {Inbeom Lee and Tongtong Jin and Bryon Aragam},
  title = {{Beyond identifiability: Learning causal representations with few environments and finite samples}},
  year = {2026},
  url = {https://arxiv.org/abs/2603.25796},
  note = {MD69: preprint; submitted 2026-03-26},
  urldate = {2026-10-03},
  eprint = {2603.25796},
  archivePrefix = {arXiv}
}

@misc{MD70,
  author = {Kwonho Kim and Heejeong Nam and Inwoo Hwang and Sanghack Lee},
  title = {{On Causal Representation Learning with Internal Auxiliaries}},
  year = {2026},
  url = {https://proceedings.mlr.press/v337/kim26e.html},
  note = {MD70: UAI 2026, PMLR 337, published},
  urldate = {2026-10-03}
}

@misc{MD71,
  author = {Manzil Zaheer and Satwik Kottur and Siamak Ravanbakhsh and Barnabas Poczos and Ruslan Salakhutdinov and Alexander Smola},
  title = {{Deep Sets}},
  year = {2017},
  url = {https://arxiv.org/abs/1703.06114},
  note = {MD71: NeurIPS 2017, published},
  urldate = {2026-10-03},
  eprint = {1703.06114},
  archivePrefix = {arXiv}
}

@misc{MD72,
  author = {Taco S. Cohen and Max Welling},
  title = {{Group Equivariant Convolutional Networks}},
  year = {2016},
  url = {https://arxiv.org/abs/1602.07576},
  note = {MD72: ICML 2016, published},
  urldate = {2026-10-03},
  eprint = {1602.07576},
  archivePrefix = {arXiv}
}

@misc{MD73,
  author = {Michael M. Bronstein and Joan Bruna and Taco Cohen and Petar Veličković},
  title = {{Geometric Deep Learning: Grids, Groups, Graphs, Geodesics, and Gauges}},
  year = {2021},
  url = {https://arxiv.org/abs/2104.13478},
  note = {MD73: arXiv monograph 2021},
  urldate = {2026-10-03},
  eprint = {2104.13478},
  archivePrefix = {arXiv}
}

@misc{MD74,
  author = {Jerome H. Friedman and Bogdan E. Popescu},
  title = {{Predictive learning via rule ensembles}},
  year = {2008},
  url = {https://arxiv.org/abs/0811.1679},
  note = {MD74: Annals of Applied Statistics 2(3), published},
  urldate = {2026-10-03},
  eprint = {0811.1679},
  archivePrefix = {arXiv}
}

@misc{MD75,
  author = {Michael Tsang and Dehua Cheng and Yan Liu},
  title = {{Detecting Statistical Interactions from Neural Network Weights}},
  year = {2018},
  url = {https://arxiv.org/abs/1705.04977},
  note = {MD75: ICLR 2018, published (preprint 2017)},
  urldate = {2026-10-03},
  eprint = {1705.04977},
  archivePrefix = {arXiv}
}

@misc{MD80,
  author = {Gauthier Gidel and Hugo Berard and Gaëtan Vignoud and Pascal Vincent and Simon Lacoste-Julien},
  title = {{A Variational Inequality Perspective on Generative Adversarial Networks}},
  year = {2019},
  url = {https://arxiv.org/abs/1802.10551},
  note = {MD80: ICLR 2019, published; arXiv初稿2018},
  urldate = {2026-10-03},
  eprint = {1802.10551},
  archivePrefix = {arXiv}
}

@misc{MD81,
  author = {Florian Schäfer and Anima Anandkumar},
  title = {{Competitive Gradient Descent}},
  year = {2019},
  url = {https://arxiv.org/abs/1905.12103},
  note = {MD81: NeurIPS 2019, published},
  urldate = {2026-10-03},
  eprint = {1905.12103},
  archivePrefix = {arXiv}
}

@misc{MD82,
  author = {Pavel Izmailov and Dmitrii Podoprikhin and Timur Garipov and Dmitry Vetrov and Andrew Gordon Wilson},
  title = {{Averaging Weights Leads to Wider Optima and Better Generalization}},
  year = {2018},
  url = {https://arxiv.org/abs/1803.05407},
  note = {MD82: UAI 2018, published},
  urldate = {2026-10-03},
  eprint = {1803.05407},
  archivePrefix = {arXiv}
}

@misc{MD83,
  author = {Wesley Maddox and Timur Garipov and Pavel Izmailov and Dmitry Vetrov and Andrew Gordon Wilson},
  title = {{A Simple Baseline for Bayesian Uncertainty in Deep Learning}},
  year = {2019},
  url = {https://arxiv.org/abs/1902.02476},
  note = {MD83: NeurIPS 2019, published},
  urldate = {2026-10-03},
  eprint = {1902.02476},
  archivePrefix = {arXiv}
}

@misc{MD84,
  author = {Karl Cobbe and Christopher Hesse and Jacob Hilton and John Schulman},
  title = {{Leveraging Procedural Generation to Benchmark Reinforcement Learning}},
  year = {2020},
  url = {https://arxiv.org/abs/1912.01588},
  note = {MD84: ICML 2020, published; arXiv初稿2019},
  urldate = {2026-10-03},
  eprint = {1912.01588},
  archivePrefix = {arXiv}
}

@misc{MD85,
  author = {Kaixin Wang and Bingyi Kang and Jie Shao and Jiashi Feng},
  title = {{Improving Generalization in Reinforcement Learning with Mixture Regularization}},
  year = {2020},
  url = {https://arxiv.org/abs/2010.10814},
  note = {MD85: NeurIPS 2020, published},
  urldate = {2026-10-03},
  eprint = {2010.10814},
  archivePrefix = {arXiv}
}

@misc{MD86,
  author = {Victor Garcia Satorras and Emiel Hoogeboom and Max Welling},
  title = {{E(n) Equivariant Graph Neural Networks}},
  year = {2021},
  url = {https://arxiv.org/abs/2102.09844},
  note = {MD86: ICML 2021, published},
  urldate = {2026-10-03},
  eprint = {2102.09844},
  archivePrefix = {arXiv}
}

@misc{MD87,
  author = {Valentin De Bortoli and James Thornton and Jeremy Heng and Arnaud Doucet},
  title = {{Diffusion Schrödinger Bridge with Applications to Score-Based Generative Modeling}},
  year = {2021},
  url = {https://arxiv.org/abs/2106.01357},
  note = {MD87: NeurIPS 2021, published},
  urldate = {2026-10-03},
  eprint = {2106.01357},
  archivePrefix = {arXiv}
}

@misc{MD88,
  author = {Aapo Hyvärinen},
  title = {{Estimation of Non-Normalized Statistical Models by Score Matching}},
  year = {2005},
  url = {https://jmlr.csail.mit.edu/papers/v6/hyvarinen05a.html},
  note = {MD88: JMLR 6(24), published},
  urldate = {2026-10-03}
}

@misc{MD89,
  author = {Yang Song and Stefano Ermon},
  title = {{Generative Modeling by Estimating Gradients of the Data Distribution}},
  year = {2019},
  url = {https://arxiv.org/abs/1907.05600},
  note = {MD89: NeurIPS 2019, published},
  urldate = {2026-10-03},
  eprint = {1907.05600},
  archivePrefix = {arXiv}
}

@misc{MD90,
  author = {Okan Koc and Alexander Soen and Chao-Kai Chiang and Masashi Sugiyama},
  title = {{Domain Adaptation and Entanglement: an Optimal Transport Perspective}},
  year = {2025},
  url = {https://proceedings.mlr.press/v258/koc25a.html},
  note = {MD90: AISTATS 2025, PMLR 258:3034–3042, published},
  urldate = {2026-10-03}
}

@misc{MD91,
  author = {Thomas Sesmat and Gabriel Meseguer-Brocal and Geoffroy Peeters},
  title = {{Where Flow Matching Leaks: Characterising the Membership Signals Along the Interpolation Path}},
  year = {2026},
  url = {https://proceedings.mlr.press/v306/sesmat26a.html},
  note = {MD91: ICML 2026（July 6–11）, PMLR 306:109269–109293, published},
  urldate = {2026-10-03}
}

@misc{MD92,
  author = {池田 思朗},
  title = {{独立成分解析の信号処理への応用}},
  year = {1999},
  url = {https://www.ism.ac.jp/~shiro/papers/etc/sice1999Jul.pdf},
  note = {MD92: 計測と制御 38(7):461–467, 1999-07-10, published},
  urldate = {2026-10-03}
}

@misc{MD93,
  author = {Robin Rombach and Andreas Blattmann and Dominik Lorenz and Patrick Esser and Björn Ommer},
  title = {{High-Resolution Image Synthesis with Latent Diffusion Models}},
  year = {2022},
  url = {https://arxiv.org/abs/2112.10752},
  note = {MD93: CVPR 2022との著者arXiv記載; arXiv初稿2021},
  urldate = {2026-10-03},
  eprint = {2112.10752},
  archivePrefix = {arXiv}
}

@misc{SY002,
  author = {Yanping Huang and Youlong Cheng and Ankur Bapna and others},
  title = {{GPipe: Efficient Training of Giant Neural Networks using Pipeline Parallelism}},
  year = {2018},
  url = {https://arxiv.org/abs/1811.06965},
  note = {SY002: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {1811.06965},
  archivePrefix = {arXiv}
}

@misc{SY004,
  author = {Yanli Zhao and Andrew Gu and Rohan Varma and others},
  title = {{PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel}},
  year = {2023},
  url = {https://arxiv.org/abs/2304.11277},
  note = {SY004: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2304.11277},
  archivePrefix = {arXiv}
}

@misc{SY005,
  author = {Wanchao Liang and Tianyu Liu and Less Wright and others},
  title = {{TorchTitan: One-stop PyTorch native solution for production ready LLM pre-training}},
  year = {2024},
  url = {https://arxiv.org/abs/2410.06511},
  note = {SY005: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2410.06511},
  archivePrefix = {arXiv}
}

@misc{SY006,
  author = {Deepak Narayanan and Mohammad Shoeybi and Jared Casper and others},
  title = {{Efficient Large-Scale Language Model Training on GPU Clusters Using Megatron-LM}},
  year = {2021},
  url = {https://arxiv.org/abs/2104.04473},
  note = {SY006: SC 2021},
  urldate = {2026-10-03},
  eprint = {2104.04473},
  archivePrefix = {arXiv}
}

@misc{SY007,
  author = {Paulius Micikevicius and Sharan Narang and Jonah Alben and others},
  title = {{Mixed Precision Training}},
  year = {2017},
  url = {https://arxiv.org/abs/1710.03740},
  note = {SY007: ICLR 2018（arXiv初出2017）},
  urldate = {2026-10-03},
  eprint = {1710.03740},
  archivePrefix = {arXiv}
}

@misc{SY008,
  author = {Tianqi Chen and Bing Xu and Chiyuan Zhang and Carlos Guestrin},
  title = {{Training Deep Nets with Sublinear Memory Cost}},
  year = {2016},
  url = {https://arxiv.org/abs/1604.06174},
  note = {SY008: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {1604.06174},
  archivePrefix = {arXiv}
}

@misc{SY010,
  author = {Gabriel Ilharco and Marco Tulio Ribeiro and Mitchell Wortsman and others},
  title = {{Editing Models with Task Arithmetic}},
  year = {2022},
  url = {https://arxiv.org/abs/2212.04089},
  note = {SY010: ICLR 2023（arXiv初出2022）},
  urldate = {2026-10-03},
  eprint = {2212.04089},
  archivePrefix = {arXiv}
}

@misc{SY011,
  author = {Prateek Yadav and Derek Tam and Leshem Choshen and Colin Raffel and Mohit Bansal},
  title = {{TIES-Merging: Resolving Interference When Merging Models}},
  year = {2023},
  url = {https://arxiv.org/abs/2306.01708},
  note = {SY011: NeurIPS 2023},
  urldate = {2026-10-03},
  eprint = {2306.01708},
  archivePrefix = {arXiv}
}

@misc{SY013,
  author = {Hao Liu and Matei Zaharia and Pieter Abbeel},
  title = {{Ring Attention with Blockwise Transformers for Near-Infinite Context}},
  year = {2023},
  url = {https://arxiv.org/abs/2310.01889},
  note = {SY013: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {2310.01889},
  archivePrefix = {arXiv}
}

@misc{SY015,
  author = {Aaron Harlap and Deepak Narayanan and Amar Phanishayee and others},
  title = {{PipeDream: Fast and Efficient Pipeline Parallel DNN Training}},
  year = {2018},
  url = {https://arxiv.org/abs/1806.03377},
  note = {SY015: arXiv版参照（公刊状況を別途確認したものは注記）},
  urldate = {2026-10-03},
  eprint = {1806.03377},
  archivePrefix = {arXiv}
}

@misc{OA001,
  author = {James Bergstra and Yoshua Bengio},
  title = {{Random Search for Hyper-Parameter Optimization}},
  year = {2012},
  url = {https://jmlr.org/papers/v13/bergstra12a.html},
  note = {OA001: JMLR 13(10), 281–305 (2012)},
  urldate = {2026-10-03}
}

@misc{OA002,
  author = {Jasper Snoek and others},
  title = {{Practical Bayesian Optimization of Machine Learning Algorithms}},
  year = {2012},
  url = {https://arxiv.org/abs/1206.2944},
  note = {OA002: arXiv preprint（会議版未再照合）},
  urldate = {2026-10-03},
  eprint = {1206.2944},
  archivePrefix = {arXiv}
}

@misc{OA003,
  author = {Yuhuai Wu and others},
  title = {{Understanding Short-Horizon Bias in Stochastic Meta-Optimization}},
  year = {2018},
  url = {https://arxiv.org/abs/1803.02021},
  note = {OA003: ICLR 2018},
  urldate = {2026-10-03},
  eprint = {1803.02021},
  archivePrefix = {arXiv}
}

@misc{OA004,
  author = {Michele Donini and others},
  title = {{Marthe: Scheduling the Learning Rate Via Online Hypergradients}},
  year = {2020},
  url = {https://www.ijcai.org/proceedings/2020/293},
  note = {OA004: IJCAI 2020, 2119–2125（この会議版を照合）},
  urldate = {2026-10-03}
}

@misc{OA005,
  author = {Tom Schaul and others},
  title = {{No More Pesky Learning Rates}},
  year = {2012},
  url = {https://arxiv.org/abs/1206.1106},
  note = {OA005: arXiv preprint（出版版未再照合）},
  urldate = {2026-10-03},
  eprint = {1206.1106},
  archivePrefix = {arXiv}
}

@misc{OA006,
  author = {Olga Wichrowska and others},
  title = {{Learned Optimizers that Scale and Generalize}},
  year = {2017},
  url = {https://arxiv.org/abs/1703.04813},
  note = {OA006: ICML 2017},
  urldate = {2026-10-03},
  eprint = {1703.04813},
  archivePrefix = {arXiv}
}

@misc{OA007,
  author = {Leslie N. Smith and others},
  title = {{Super-Convergence: Very Fast Training of Neural Networks Using Large Learning Rates}},
  year = {2017},
  url = {https://arxiv.org/abs/1708.07120},
  note = {OA007: arXiv preprint},
  urldate = {2026-10-03},
  eprint = {1708.07120},
  archivePrefix = {arXiv}
}

@misc{OA008,
  author = {Robin M. Schmidt and others},
  title = {{Descending through a Crowded Valley - Benchmarking Deep Learning Optimizers}},
  year = {2020},
  url = {https://arxiv.org/abs/2007.01547},
  note = {OA008: arXiv preprint（会議版未再照合）},
  urldate = {2026-10-03},
  eprint = {2007.01547},
  archivePrefix = {arXiv}
}

@misc{OA009,
  author = {Yikang Shen and others},
  title = {{Power Scheduler: A Batch Size and Token Number Agnostic Learning Rate Scheduler}},
  year = {2024},
  url = {https://arxiv.org/abs/2408.13359v1},
  note = {OA009: arXiv preprint（元リンクのv1を確認）},
  urldate = {2026-10-03},
  eprint = {2408.13359},
  archivePrefix = {arXiv}
}

@misc{OA010,
  author = {Zhao Song and others},
  title = {{An Automatic Learning Rate Schedule Algorithm for Achieving Faster Convergence and Steeper Descent}},
  year = {2023},
  url = {https://arxiv.org/abs/2310.11291},
  note = {OA010: arXiv preprint},
  urldate = {2026-10-03},
  eprint = {2310.11291},
  archivePrefix = {arXiv}
}

@misc{OA011,
  author = {Yuchen Jin and others},
  title = {{AutoLRS: Automatic Learning-Rate Schedule by Bayesian Optimization on the Fly}},
  year = {2021},
  url = {https://arxiv.org/abs/2105.10762},
  note = {OA011: ICLR 2021},
  urldate = {2026-10-03},
  eprint = {2105.10762},
  archivePrefix = {arXiv}
}

@misc{OA012,
  author = {Xingyu Xie and others},
  title = {{Optimization Hyper-parameter Laws for Large Language Models}},
  year = {2024},
  url = {https://arxiv.org/abs/2409.04777v1},
  note = {OA012: arXiv preprint（元リンクのv1を確認、最新版は2026改訂）},
  urldate = {2026-10-03},
  eprint = {2409.04777},
  archivePrefix = {arXiv}
}

@misc{OA013,
  author = {Kairong Luo and others},
  title = {{A Multi-Power Law for Loss Curve Prediction Across Learning Rate Schedules}},
  year = {2025},
  url = {https://arxiv.org/abs/2503.12811},
  note = {OA013: arXiv preprint},
  urldate = {2026-10-03},
  eprint = {2503.12811},
  archivePrefix = {arXiv}
}

@misc{OA014,
  author = {Yuki Tsukada and others},
  title = {{Relationship between Batch Size and Number of Steps Needed for Nonconvex Optimization of Stochastic Gradient Descent using Armijo Line Search}},
  year = {2023},
  url = {https://arxiv.org/abs/2307.13831},
  note = {OA014: arXiv preprint},
  urldate = {2026-10-03},
  eprint = {2307.13831},
  archivePrefix = {arXiv}
}

@misc{OA015,
  author = {Tao Zhang and others},
  title = {{kDecay: Just adding k-decay items on Learning-Rate Schedule to improve Neural Networks}},
  year = {2020},
  url = {https://arxiv.org/abs/2004.05909},
  note = {OA015: arXiv preprint},
  urldate = {2026-10-03},
  eprint = {2004.05909},
  archivePrefix = {arXiv}
}

@misc{OA016,
  author = {Prabhu Teja Sivaprasad and others},
  title = {{Optimizer Benchmarking Needs to Account for Hyperparameter Tuning}},
  year = {2019},
  url = {https://arxiv.org/abs/1910.11758},
  note = {OA016: ICML 2020},
  urldate = {2026-10-03},
  eprint = {1910.11758},
  archivePrefix = {arXiv}
}

@misc{OA017,
  author = {Zixiang Chen and others},
  title = {{Why Does Sharpness-Aware Minimization Generalize Better Than SGD?}},
  year = {2023},
  url = {https://arxiv.org/abs/2310.07269},
  note = {OA017: NeurIPS 2023},
  urldate = {2026-10-03},
  eprint = {2310.07269},
  archivePrefix = {arXiv}
}

@misc{OA018,
  author = {Szilvia Ujváry and others},
  title = {{Rethinking Sharpness-Aware Minimization as Variational Inference}},
  year = {2022},
  url = {https://arxiv.org/abs/2210.10452},
  note = {OA018: arXiv preprint},
  urldate = {2026-10-03},
  eprint = {2210.10452},
  archivePrefix = {arXiv}
}

@misc{OA019,
  author = {Thomas Möllenhoff and others},
  title = {{SAM as an Optimal Relaxation of Bayes}},
  year = {2022},
  url = {https://arxiv.org/abs/2210.01620},
  note = {OA019: ICLR 2023},
  urldate = {2026-10-03},
  eprint = {2210.01620},
  archivePrefix = {arXiv}
}

@misc{OA020,
  author = {Jean Kaddour and others},
  title = {{When Do Flat Minima Optimizers Work?}},
  year = {2022},
  url = {https://arxiv.org/abs/2202.00661},
  note = {OA020: NeurIPS 2022},
  urldate = {2026-10-03},
  eprint = {2202.00661},
  archivePrefix = {arXiv}
}

@misc{OA021,
  author = {Simran Kaur and others},
  title = {{On the Maximum Hessian Eigenvalue and Generalization}},
  year = {2022},
  url = {https://arxiv.org/abs/2206.10654},
  note = {OA021: NeurIPS 2022 workshop; PMLR 187, 51–65 (2023)},
  urldate = {2026-10-03},
  eprint = {2206.10654},
  archivePrefix = {arXiv}
}

@misc{OA022,
  author = {Sidak Pal Singh and others},
  title = {{The Hessian perspective into the Nature of Convolutional Neural Networks}},
  year = {2023},
  url = {https://arxiv.org/abs/2305.09088},
  note = {OA022: ICML 2023},
  urldate = {2026-10-03},
  eprint = {2305.09088},
  archivePrefix = {arXiv}
}

@misc{OA023,
  author = {Harsh Rangwani and others},
  title = {{Escaping Saddle Points for Effective Generalization on Class-Imbalanced Data}},
  year = {2022},
  url = {https://arxiv.org/abs/2212.13827},
  note = {OA023: NeurIPS 2022},
  urldate = {2026-10-03},
  eprint = {2212.13827},
  archivePrefix = {arXiv}
}

@misc{OA024,
  author = {Georgios Arvanitidis and others},
  title = {{Latent Space Oddity: on the Curvature of Deep Generative Models}},
  year = {2017},
  url = {https://arxiv.org/abs/1710.11379},
  note = {OA024: ICLR 2018},
  urldate = {2026-10-03},
  eprint = {1710.11379},
  archivePrefix = {arXiv}
}

@misc{OA025,
  author = {Hongyang R. Zhang and others},
  title = {{Noise Stability Optimization for Finding Flat Minima: A Hessian-based Regularization Approach}},
  year = {2023},
  url = {https://arxiv.org/abs/2306.08553},
  note = {OA025: arXiv preprint（2024年改訂題名）},
  urldate = {2026-10-03},
  eprint = {2306.08553},
  archivePrefix = {arXiv}
}

@misc{OA026,
  author = {Kayhan Behdin and others},
  title = {{Improved Deep Neural Network Generalization Using m-Sharpness-Aware Minimization}},
  year = {2022},
  url = {https://arxiv.org/abs/2212.04343},
  note = {OA026: arXiv preprint},
  urldate = {2026-10-03},
  eprint = {2212.04343},
  archivePrefix = {arXiv}
}

@misc{OA027,
  author = {Peng Mi and others},
  title = {{Make Sharpness-Aware Minimization Stronger: A Sparsified Perturbation Approach}},
  year = {2022},
  url = {https://arxiv.org/abs/2210.05177},
  note = {OA027: NeurIPS 2022},
  urldate = {2026-10-03},
  eprint = {2210.05177},
  archivePrefix = {arXiv}
}

@misc{OA028,
  author = {Tom Sherborne and others},
  title = {{TRAM: Bridging Trust Regions and Sharpness Aware Minimization}},
  year = {2023},
  url = {https://arxiv.org/abs/2310.03646},
  note = {OA028: ICLR 2024 spotlight},
  urldate = {2026-10-03},
  eprint = {2310.03646},
  archivePrefix = {arXiv}
}

@misc{OA029,
  author = {Yong Liu and others},
  title = {{Random Sharpness-Aware Minimization}},
  year = {2022},
  url = {https://papers.nips.cc/paper_files/paper/2022/hash/9b79416c0dc4b09feaa169ed5cdd63d4-Abstract-Conference.html},
  note = {OA029: NeurIPS 2022},
  urldate = {2026-10-03}
}

@misc{OA030,
  author = {Yang Zhao and others},
  title = {{Randomized Sharpness-Aware Training for Boosting Computational Efficiency in Deep Learning}},
  year = {2022},
  url = {https://arxiv.org/abs/2203.09962},
  note = {OA030: arXiv preprint},
  urldate = {2026-10-03},
  eprint = {2203.09962},
  archivePrefix = {arXiv}
}

@misc{OA031,
  author = {Yong Liu and others},
  title = {{Towards Efficient and Scalable Sharpness-Aware Minimization}},
  year = {2022},
  url = {https://arxiv.org/abs/2203.02714},
  note = {OA031: CVPR 2022},
  urldate = {2026-10-03},
  eprint = {2203.02714},
  archivePrefix = {arXiv}
}

@misc{OA032,
  author = {Hao Sun and others},
  title = {{AdaSAM: Boosting Sharpness-Aware Minimization with Adaptive Learning Rate and Momentum for Training Deep Neural Networks}},
  year = {2023},
  url = {https://arxiv.org/abs/2303.00565},
  note = {OA032: arXiv preprint},
  urldate = {2026-10-03},
  eprint = {2303.00565},
  archivePrefix = {arXiv}
}

@misc{OA033,
  author = {Xiangning Chen and others},
  title = {{When Vision Transformers Outperform ResNets without Pre-training or Strong Data Augmentations}},
  year = {2021},
  url = {https://arxiv.org/abs/2106.01548},
  note = {OA033: ICLR 2022 spotlight},
  urldate = {2026-10-03},
  eprint = {2106.01548},
  archivePrefix = {arXiv}
}

@misc{OA034,
  author = {Qihuang Zhong and others},
  title = {{Improving Sharpness-Aware Minimization with Fisher Mask for Better Generalization on Language Models}},
  year = {2022},
  url = {https://aclanthology.org/2022.findings-emnlp.300/},
  note = {OA034: Findings of EMNLP 2022, 4064–4085},
  urldate = {2026-10-03}
}

@misc{OA035,
  author = {Rajhans Singh and others},
  title = {{Improving Shape Awareness and Interpretability in Deep Networks Using Geometric Moments}},
  year = {2022},
  url = {https://arxiv.org/abs/2205.11722},
  note = {OA035: CVPR 2023 Workshop: Deep Learning for Geometric Computing},
  urldate = {2026-10-03},
  eprint = {2205.11722},
  archivePrefix = {arXiv}
}

@misc{OA036,
  author = {Feng Chen and others},
  title = {{Stochastic Collapse: How Gradient Noise Attracts SGD Dynamics Towards Simpler Subnetworks}},
  year = {2023},
  url = {https://arxiv.org/abs/2306.04251},
  note = {OA036: NeurIPS 2023},
  urldate = {2026-10-03},
  eprint = {2306.04251},
  archivePrefix = {arXiv}
}

@misc{OA037,
  author = {Florian Seligmann and others},
  title = {{Beyond Deep Ensembles: A Large-Scale Evaluation of Bayesian Deep Learning under Distribution Shift}},
  year = {2023},
  url = {https://arxiv.org/abs/2306.12306},
  note = {OA037: arXiv preprint（出版版未再照合）},
  urldate = {2026-10-03},
  eprint = {2306.12306},
  archivePrefix = {arXiv}
}

@misc{OA039,
  author = {Li Wang and others},
  title = {{The Implicit Regularization of Momentum Gradient Descent with Early Stopping}},
  year = {2022},
  url = {https://arxiv.org/abs/2201.05405},
  note = {OA039: arXiv preprint},
  urldate = {2026-10-03},
  eprint = {2201.05405},
  archivePrefix = {arXiv}
}

@misc{OB001,
  author = {James Martens},
  title = {{New Insights and Perspectives on the Natural Gradient Method}},
  year = {2020},
  url = {https://jmlr.org/papers/v21/17-678.html},
  note = {OB001: JMLR 21(146):1–76, 2020, published},
  urldate = {2026-10-03}
}

@misc{OB002,
  author = {Christopher J. Shallue and Jaehoon Lee and Joseph Antognini and Jascha Sohl-Dickstein and Roy Frostig and George E. Dahl},
  title = {{Measuring the Effects of Data Parallelism on Neural Network Training}},
  year = {2019},
  url = {https://jmlr.org/papers/v20/18-789.html},
  note = {OB002: JMLR 20(112):1–49, 2019, published},
  urldate = {2026-10-03}
}

@misc{OB003,
  author = {Shuai Zheng and Haibin Lin and Sheng Zha and Mu Li},
  title = {{Accelerated Large Batch Optimization of BERT Pretraining in 54 minutes}},
  year = {2020},
  url = {https://arxiv.org/abs/2006.13484},
  note = {OB003: technical report, 2020; arXiv明記で査読venueなし},
  urldate = {2026-10-03},
  eprint = {2006.13484},
  archivePrefix = {arXiv}
}

@misc{OB004,
  author = {Raghu Bollapragada and Richard Byrd and Jorge Nocedal},
  title = {{Adaptive Sampling Strategies for Stochastic Optimization}},
  year = {2018},
  url = {https://arxiv.org/abs/1710.11258},
  note = {OB004: SIAM Journal on Optimization, 2018, published; arXiv初稿2017},
  urldate = {2026-10-03},
  eprint = {1710.11258},
  archivePrefix = {arXiv}
}

@misc{OB005,
  author = {Tim Tsz-Kit Lau and Han Liu and Mladen Kolar},
  title = {{AdAdaGrad: Adaptive Batch Size Schemes for Adaptive Gradient Methods}},
  year = {2026},
  url = {https://arxiv.org/abs/2402.11215},
  note = {OB005: Statistical Learning and Data Science, online 2026-08-11, in press / journal pre-proof（出版社検索取得書誌）; arXiv初稿2024、v4 2026-08-25},
  urldate = {2026-10-03},
  eprint = {2402.11215},
  archivePrefix = {arXiv}
}

@misc{OB006,
  author = {Zhengda Bian and Shenggui Li and Wei Wang and Yang You},
  title = {{Online Evolutionary Batch Size Orchestration for Scheduling Deep Learning Workloads in GPU Clusters}},
  year = {2021},
  url = {https://arxiv.org/abs/2108.03645},
  note = {OB006: SC 2021 acceptedとの著者arXiv記載;初稿2021},
  urldate = {2026-10-03},
  eprint = {2108.03645},
  archivePrefix = {arXiv}
}

@misc{OB007,
  author = {Xiaoxin He and Fuzhao Xue and Xiaozhe Ren and Yang You},
  title = {{Large-Scale Deep Learning Optimizations: A Comprehensive Survey}},
  year = {2021},
  url = {https://arxiv.org/abs/2111.00856},
  note = {OB007: arXiv survey公開2021;正式刊行版未照合},
  urldate = {2026-10-03},
  eprint = {2111.00856},
  archivePrefix = {arXiv}
}

@misc{OB008,
  author = {Lukas Balles and Fabian Pedregosa and Nicolas Le Roux},
  title = {{The Geometry of Sign Gradient Descent}},
  year = {2020},
  url = {https://arxiv.org/abs/2002.08056},
  note = {OB008: arXiv公開2020;正式刊行版未照合},
  urldate = {2026-10-03},
  eprint = {2002.08056},
  archivePrefix = {arXiv}
}

@misc{OB009,
  author = {Róisín Luo and James McDermott and Christian Gagné and Qiang Sun and Colm O'Riordan},
  title = {{Optimization-Induced Dynamics of Lipschitz Continuity in Neural Networks}},
  year = {2025},
  url = {https://arxiv.org/abs/2506.18588},
  note = {OB009: preprint; submitted 2025-06-23, revised 2025-11-14},
  urldate = {2026-10-03},
  eprint = {2506.18588},
  archivePrefix = {arXiv}
}

@misc{OB010,
  author = {Ke Liang Xiao and Noah Marshall and Atish Agarwala and Elliot Paquette},
  title = {{Exact Risk Curves of signSGD in High-Dimensions: Quantifying Preconditioning and Noise-Compression Effects}},
  year = {2025},
  url = {https://arxiv.org/abs/2411.12135},
  note = {OB010: ICML 2025, PMLR 267:68391–68439; arXiv初稿2024、v3 2026-03-25},
  urldate = {2026-10-03},
  eprint = {2411.12135},
  archivePrefix = {arXiv}
}

@misc{OB011,
  author = {Shikai Qiu and Lechao Xiao and Andrew Gordon Wilson and Jeffrey Pennington and Atish Agarwala},
  title = {{Scaling Collapse Reveals Universal Dynamics in Compute-Optimally Trained Neural Networks}},
  year = {2025},
  url = {https://proceedings.mlr.press/v267/qiu25j.html},
  note = {OB011: ICML 2025, PMLR 267:50697–50720, published},
  urldate = {2026-10-03}
}

@misc{OB012,
  author = {Xiangru Lian and Ce Zhang and Huan Zhang and Cho-Jui Hsieh and Wei Zhang and Ji Liu},
  title = {{Can Decentralized Algorithms Outperform Centralized Algorithms? A Case Study for Decentralized Parallel Stochastic Gradient Descent}},
  year = {2017},
  url = {https://arxiv.org/abs/1705.09056},
  note = {OB012: NeurIPS 2017, published},
  urldate = {2026-10-03},
  eprint = {1705.09056},
  archivePrefix = {arXiv}
}

@misc{OB013,
  author = {Tao Lin and Lingjing Kong and Sebastian U. Stich and Martin Jaggi},
  title = {{Extrapolation for Large-batch Training in Deep Learning}},
  year = {2020},
  url = {https://arxiv.org/abs/2006.05720},
  note = {OB013: ICML 2020, PMLR 119:6094–6104, published},
  urldate = {2026-10-03},
  eprint = {2006.05720},
  archivePrefix = {arXiv}
}

@misc{OB014,
  author = {Michael Diskin and Alexey Bukhtiyarov and Max Ryabinin and Lucile Saulnier and Quentin Lhoest and Anton Sinitsin and Dmitry Popov and Dmitry Pyrkin and Maxim Kashirin and Alexander Borzunov and Albert Villanova del Moral and Denis Mazur and Ilia Kobelev and Yacine Jernite and Thomas Wolf and Gennady Pekhimenko},
  title = {{Distributed Deep Learning in Open Collaborations}},
  year = {2021},
  url = {https://arxiv.org/abs/2106.10207},
  note = {OB014: NeurIPS 2021 acceptedとの著者arXiv記載},
  urldate = {2026-10-03},
  eprint = {2106.10207},
  archivePrefix = {arXiv}
}

@misc{OB015,
  author = {Sohom Mukherjee and Nicolas Loizou and Sebastian U. Stich},
  title = {{Locally Adaptive Federated Learning}},
  year = {2023},
  url = {https://arxiv.org/abs/2307.06306},
  note = {OB015: arXiv初稿2023、v2 2024-05-14;正式刊行版未照合},
  urldate = {2026-10-03},
  eprint = {2307.06306},
  archivePrefix = {arXiv}
}

@misc{OB016,
  author = {Belhal Karimi and Ping Li and Xiaoyun Li},
  title = {{Layer-wise and Dimension-wise Locally Adaptive Federated Learning}},
  year = {2023},
  url = {https://arxiv.org/abs/2110.00532},
  note = {OB016: UAI 2023, PMLR 216:1037–1046; arXiv初稿2021、刊行題名はFed-LAMBで始まる},
  urldate = {2026-10-03},
  eprint = {2110.00532},
  archivePrefix = {arXiv}
}

@misc{OB017,
  author = {John Nguyen and Kshitiz Malik and Hongyuan Zhan and Ashkan Yousefpour and Michael Rabbat and Mani Malek and Dzmitry Huba},
  title = {{Federated Learning with Buffered Asynchronous Aggregation}},
  year = {2022},
  url = {https://arxiv.org/abs/2106.06639},
  note = {OB017: AISTATS 2022, PMLR 151:3581–3607, published; arXiv初稿2021},
  urldate = {2026-10-03},
  eprint = {2106.06639},
  archivePrefix = {arXiv}
}

@misc{OB018,
  author = {Anastasia Koloskova and Sebastian U. Stich and Martin Jaggi},
  title = {{Decentralized Stochastic Optimization and Gossip Algorithms with Compressed Communication}},
  year = {2019},
  url = {https://arxiv.org/abs/1902.00340},
  note = {OB018: ICML 2019, PMLR 97:3478–3487, published;記載URLはarXiv初稿},
  urldate = {2026-10-03},
  eprint = {1902.00340},
  archivePrefix = {arXiv}
}

@misc{OB019,
  author = {Kai Chen and Qiang Huo},
  title = {{Scalable Training of Deep Learning Machines by Incremental Block Training with Intra-block Parallel Optimization and Blockwise Model-Update Filtering}},
  year = {2016},
  url = {https://www.microsoft.com/en-us/research/publication/scalable-training-deep-learning-machines-incremental-block-training-intra-block-parallel-optimization-blockwise-model-update-filtering/},
  note = {OB019: ICASSP 2016, March 2016;著者所属機関の公開書誌・要旨確認},
  urldate = {2026-10-03}
}

@misc{OB020,
  author = {Depen Morwani and Itai Shapira and Nikhil Vyas and Eran Malach and Sham Kakade and Lucas Janson},
  title = {{A New Perspective on Shampoo's Preconditioner}},
  year = {2025},
  url = {https://arxiv.org/abs/2406.17748},
  note = {OB020: ICLR 2025, published; arXiv初稿2024},
  urldate = {2026-10-03},
  eprint = {2406.17748},
  archivePrefix = {arXiv}
}

@misc{OB021,
  author = {Fan Bao and Guoqiang Wu and Chongxuan Li and Jun Zhu and Bo Zhang},
  title = {{Stability and Generalization of Bilevel Programming in Hyperparameter Optimization}},
  year = {2021},
  url = {https://arxiv.org/abs/2106.04188},
  note = {OB021: NeurIPS 2021, published},
  urldate = {2026-10-03},
  eprint = {2106.04188},
  archivePrefix = {arXiv}
}

@misc{OB022,
  author = {Hanxiao Liu and Karen Simonyan and Yiming Yang},
  title = {{DARTS: Differentiable Architecture Search}},
  year = {2019},
  url = {https://arxiv.org/abs/1806.09055},
  note = {OB022: ICLR 2019 publishedとの著者arXiv記載;初稿2018},
  urldate = {2026-10-03},
  eprint = {1806.09055},
  archivePrefix = {arXiv}
}

@misc{OB023,
  author = {Thomas M. Moerland and Joost Broekens and Aske Plaat and Catholijn M. Jonker},
  title = {{Model-based Reinforcement Learning: A Survey}},
  year = {2023},
  url = {https://arxiv.org/abs/2006.16712},
  note = {OB023: Foundations and Trends in Machine Learning 16(1):1–118, 2023（出版社公開書誌確認）; arXiv初稿2020},
  urldate = {2026-10-03},
  eprint = {2006.16712},
  archivePrefix = {arXiv}
}

@misc{OB024,
  author = {Dongsung Huh and Avinash Baidya},
  title = {{The Missing Invariance Principle Found -- the Reciprocal Twin of Invariant Risk Minimization}},
  year = {2022},
  url = {https://arxiv.org/abs/2205.14546},
  note = {OB024: NeurIPS 2022との著者arXiv記載},
  urldate = {2026-10-03},
  eprint = {2205.14546},
  archivePrefix = {arXiv}
}

@misc{OB025,
  author = {Yihua Zhang and Pranay Sharma and Parikshit Ram and Mingyi Hong and Kush Varshney and Sijia Liu},
  title = {{What Is Missing in IRM Training and Evaluation? Challenges and Solutions}},
  year = {2023},
  url = {https://arxiv.org/abs/2303.02343},
  note = {OB025: ICLR 2023 acceptedとの著者arXiv記載},
  urldate = {2026-10-03},
  eprint = {2303.02343},
  archivePrefix = {arXiv}
}

@misc{OB026,
  author = {Ryo Sato and Mirai Tanaka and Akiko Takeda},
  title = {{A Gradient Method for Multilevel Optimization}},
  year = {2021},
  url = {https://arxiv.org/abs/2105.13954},
  note = {OB026: NeurIPS 2021 camera-readyとの著者arXiv記載},
  urldate = {2026-10-03},
  eprint = {2105.13954},
  archivePrefix = {arXiv}
}

@misc{OB027,
  author = {本川 哲哉 and 手塚 太郎},
  title = {{ニューラルネットワークにおける適応的二次最適化手法}},
  year = {2019},
  url = {https://db-event.jpn.org/deim2019/post/papers/349.pdf},
  note = {OB027: DEIM Forum 2019, A4-2, 公開研究会論文},
  urldate = {2026-10-03}
}

@misc{OB028,
  author = {矢部 博},
  title = {{共役勾配法}},
  year = {1987},
  url = {https://orsj.org/wp-content/or-archives50/pdf/bul/Vol.32_06_363.pdf},
  note = {OB028: オペレーションズ・リサーチ 32(6), pp.363–367, 1987, 学会解説},
  urldate = {2026-10-03}
}

@misc{OB029,
  author = {八巻 直一 and 矢部 博},
  title = {{非線形計画法（3）—無制約最適化問題—}},
  year = {1995},
  url = {https://orsj.org/wp-content/or-archives50/pdf/bul/Vol.40_01_055.pdf},
  note = {OB029: オペレーションズ・リサーチ 40(1), pp.55–60, 1995, 学会解説},
  urldate = {2026-10-03}
}
