Smarter by the Moment: Environment-Driven Dynamic Policies for Continual LLM Improvement
TL;DRLets an LLM turn its own history into strategies — continual gains, no retraining.
BibTeX
@inproceedings{chang2026smarter,
title = {Smarter by the Moment: Environment-Driven Dynamic Policies for Continual {LLM} Improvement},
author = {Chang, Ting-Wei and Chen, Po-Chun and Huang, Hen-Hsen and Chen, Hsin-Hsi},
booktitle = {Proceedings of the Conference on Language Modeling ({COLM})},
year = {2026},
eprint = {2609.16800},
archivePrefix = {arXiv},
primaryClass = {cs.CL},
doi = {10.48550/arXiv.2609.16800},
url = {https://arxiv.org/abs/2609.16800}
}