% Jev Survey bibliography — generated by scripts/build.py from data/papers.json. Do not edit.
% 71 records: 17 core, 1 peripheral, 53 background. Cutoff 2026-09-24.

@misc{arxiv2609_22753,
  title = {{Replacing Large Language Models with Jev Decision Models for Low-Latency Edge Service Orchestration}},
  author = {Delong Li and Xu Wang and Haochen Gong and Rui Lang and Guangsheng Yu},
  year = {2026},
  eprint = {2609.22753},
  archivePrefix = {arXiv},
  primaryClass = {cs.DC},
  url = {https://arxiv.org/abs/2609.22753v1}
}

@misc{arxiv2609_23136,
  title = {{Fast Intent-Driven Service Orchestration with Jev for 6G Edge Networks}},
  author = {Delong Li and Xu Wang and Haochen Gong and Rui Lang and Guangsheng Yu},
  year = {2026},
  eprint = {2609.23136},
  archivePrefix = {arXiv},
  primaryClass = {cs.NI},
  url = {https://arxiv.org/abs/2609.23136v1}
}

@misc{arxiv2609_23886,
  title = {{this-that-model-1.0: A typed decision model that decides in 30 ms, for a millionth of a cent}},
  author = {Zehua Cheng and Wei Dai and Jiahao Sun},
  year = {2026},
  eprint = {2609.23886},
  archivePrefix = {arXiv},
  primaryClass = {cs.CL},
  url = {https://arxiv.org/abs/2609.23886v1}
}

@misc{arxiv2609_23959,
  title = {{Open-Jev Judgments on CallScreenBench: Calibrated One-Pass Scam Screening with a Small Language Model}},
  author = {Simiao Ren and Kidus Zewde and Xingyu Shen and Yuchen Zhou and Dennis Ng and Ankit Raj and Tommy Duong and Yuxin Zhang and Neo Tiangratanakul},
  year = {2026},
  eprint = {2609.23959},
  archivePrefix = {arXiv},
  primaryClass = {cs.CL},
  url = {https://arxiv.org/abs/2609.23959v1}
}

@misc{arxiv2609_23986,
  title = {{Jev-Mem: System-One-Controlled Agentic Memory for Efficient AI Agents}},
  author = {Dongming Jiang and Yi Li and Bingzhe Li},
  year = {2026},
  eprint = {2609.23986},
  archivePrefix = {arXiv},
  primaryClass = {cs.AI},
  url = {https://arxiv.org/abs/2609.23986v1}
}

@misc{arxiv2609_24052,
  title = {{Calibrated Decisions at Scale: Converting Police Crash Narratives into Probabilistic Crash Variables with a System One Model (Jev)}},
  author = {Amir Rafe and Subasish Das},
  year = {2026},
  eprint = {2609.24052},
  archivePrefix = {arXiv},
  primaryClass = {cs.CL},
  url = {https://arxiv.org/abs/2609.24052v1}
}

@misc{arxiv2609_24395,
  title = {{JEVQA - Video Quality from Metadata, Bitstream, and Pixel Features with a General-Purpose Decision Model}},
  author = {Werner Robitza},
  year = {2026},
  eprint = {2609.24395},
  archivePrefix = {arXiv},
  primaryClass = {eess.IV},
  url = {https://arxiv.org/abs/2609.24395v1}
}

@misc{arxiv2609_24574,
  title = {{Evaluating Decision Models for Text Annotation in Computational Social Science}},
  author = {Hazem Ibrahim and Yasir Zaki},
  year = {2026},
  eprint = {2609.24574},
  archivePrefix = {arXiv},
  primaryClass = {cs.CL},
  url = {https://arxiv.org/abs/2609.24574v1}
}

@misc{arxiv2609_24965,
  title = {{Jev for Scientific Decisions: Evaluating Semantic Choices and Their Consequences}},
  author = {Boyuan Deng and Shuyi Fan and Hongyang Zhang and Xinhong Xie},
  year = {2026},
  eprint = {2609.24965},
  archivePrefix = {arXiv},
  primaryClass = {cs.CL},
  url = {https://arxiv.org/abs/2609.24965v1}
}

@misc{arxiv2609_25845,
  title = {{Visual Jev: Accurate and Efficient Decisions from Shared Visual Context}},
  author = {Guanxu Yu and Yuhang Yao},
  year = {2026},
  eprint = {2609.25845},
  archivePrefix = {arXiv},
  primaryClass = {cs.CV},
  url = {https://arxiv.org/abs/2609.25845v1}
}

@misc{arxiv2609_26532,
  title = {{REFLEX with Jev for Efficient Selective Control in LLM Agents}},
  author = {Tiantong Wu and Wei Yang Bryan Lim},
  year = {2026},
  eprint = {2609.26532},
  archivePrefix = {arXiv},
  primaryClass = {cs.AI},
  url = {https://arxiv.org/abs/2609.26532v1}
}

@misc{arxiv2609_26550,
  title = {{JEV-as-a-Judge: Accept When Confident, Escalate When Unsure}},
  author = {Yubo Li and Yidi Miao and Ramayya Krishnan and Rema Padman},
  year = {2026},
  eprint = {2609.26550},
  archivePrefix = {arXiv},
  primaryClass = {cs.AI},
  url = {https://arxiv.org/abs/2609.26550v1}
}

@misc{arxiv2609_26758,
  title = {{Type-Safe Is Not Error-Free: A Constrained Decision Head Follows the Option Name, Not the Rubric Bound to It}},
  author = {Yu Sun and Junhao Xu},
  year = {2026},
  eprint = {2609.26758},
  archivePrefix = {arXiv},
  primaryClass = {cs.AI},
  url = {https://arxiv.org/abs/2609.26758v1}
}

@misc{arxiv2609_27331,
  title = {{JEV-Star: Fast, Low-Cost StarCraft II Control with Language-Model Planning}},
  author = {Weiyu Ma and Liangbing Zhao and Yongcheng Zeng and Jian Zhao},
  year = {2026},
  eprint = {2609.27331},
  archivePrefix = {arXiv},
  primaryClass = {cs.GT},
  url = {https://arxiv.org/abs/2609.27331v1}
}

@misc{arxiv2609_27535,
  title = {{KITE: Scaling Jev Population Experiments with Sparse Flagship Calibration}},
  author = {Hengyu Li},
  year = {2026},
  eprint = {2609.27535},
  archivePrefix = {arXiv},
  primaryClass = {cs.MA},
  url = {https://arxiv.org/abs/2609.27535v1}
}

@misc{arxiv2609_27607,
  title = {{Can Jev Judge Radiology Reports? Evaluating a System One Model for Clinical Factuality}},
  author = {Jiaju Huang and Hao Yang and Xinyu Ma and Xinglong Liang and Kunyan Cai and Junqiang Ma and Shaobin Chen and Yue Sun and Tao Tan},
  year = {2026},
  eprint = {2609.27607},
  archivePrefix = {arXiv},
  primaryClass = {cs.CL},
  url = {https://arxiv.org/abs/2609.27607v1}
}

@misc{arxiv2609_27678,
  title = {{Same Scores, Different Decisions: Evaluating JEV and Language Models for Legal Document Understanding}},
  author = {Fan Zhang and Yankai Chen and Zhuohan Xie and Yixi Zhou and Sijia Peng and Lei Fan and Xinhua Ji and Cunyuan Zheng and Huangyong Shan and Philip S. Yu and Xue Liu and Yu Chen and Preslav Nakov and Songwei He},
  year = {2026},
  eprint = {2609.27678},
  archivePrefix = {arXiv},
  primaryClass = {cs.CL},
  url = {https://arxiv.org/abs/2609.27678v1}
}

@misc{arxiv2609_25498,
  title = {{Universal Fractal Natural Language Decision Map: Real-Time Edge Triage Across Heterogeneous Domains}},
  author = {Volkan Dağlı and Zerrin Dağlı and Dağhan Dağlı},
  year = {2026},
  eprint = {2609.25498},
  archivePrefix = {arXiv},
  primaryClass = {cs.NE},
  url = {https://arxiv.org/abs/2609.25498v1},
  doi = {10.5281/zenodo.22867426}
}

@misc{arxiv1705_08500,
  title = {{Selective Classification for Deep Neural Networks}},
  author = {Yonatan Geifman and Ran El-Yaniv},
  year = {2017},
  eprint = {1705.08500},
  archivePrefix = {arXiv},
  primaryClass = {cs.LG},
  url = {https://arxiv.org/abs/1705.08500v2}
}

@misc{arxiv1706_04599,
  title = {{On Calibration of Modern Neural Networks}},
  author = {Chuan Guo and Geoff Pleiss and Yu Sun and Kilian Q. Weinberger},
  year = {2017},
  eprint = {1706.04599},
  archivePrefix = {arXiv},
  primaryClass = {cs.LG},
  url = {https://arxiv.org/abs/1706.04599v2},
  note = {ICML 2017}
}

@misc{arxiv1810_04805,
  title = {{BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding}},
  author = {Jacob Devlin and Ming-Wei Chang and Kenton Lee and Kristina Toutanova},
  year = {2018},
  eprint = {1810.04805},
  archivePrefix = {arXiv},
  primaryClass = {cs.CL},
  url = {https://arxiv.org/abs/1810.04805v2}
}

@misc{arxiv1901_09192,
  title = {{SelectiveNet: A Deep Neural Network with an Integrated Reject Option}},
  author = {Yonatan Geifman and Ran El-Yaniv},
  year = {2019},
  eprint = {1901.09192},
  archivePrefix = {arXiv},
  primaryClass = {cs.LG},
  url = {https://arxiv.org/abs/1901.09192v4},
  note = {ICML 2019}
}

@misc{arxiv1906_02530,
  title = {{Can You Trust Your Model's Uncertainty? Evaluating Predictive Uncertainty Under Dataset Shift}},
  author = {Yaniv Ovadia and Emily Fertig and Jie Ren and Zachary Nado and D Sculley and Sebastian Nowozin and Joshua V. Dillon and Balaji Lakshminarayanan and Jasper Snoek},
  year = {2019},
  eprint = {1906.02530},
  archivePrefix = {arXiv},
  primaryClass = {stat.ML},
  url = {https://arxiv.org/abs/1906.02530v2},
  note = {NeurIPS 2019}
}

@misc{arxiv1909_00161,
  title = {{Benchmarking Zero-shot Text Classification: Datasets, Evaluation and Entailment Approach}},
  author = {Wenpeng Yin and Jamaal Hay and Dan Roth},
  year = {2019},
  eprint = {1909.00161},
  archivePrefix = {arXiv},
  primaryClass = {cs.CL},
  url = {https://arxiv.org/abs/1909.00161v1},
  note = {EMNLP 2019}
}

@misc{arxiv2006_01862,
  title = {{Consistent Estimators for Learning to Defer to an Expert}},
  author = {Hussein Mozannar and David Sontag},
  year = {2020},
  eprint = {2006.01862},
  archivePrefix = {arXiv},
  primaryClass = {cs.LG},
  url = {https://arxiv.org/abs/2006.01862v3},
  note = {ICML 2020}
}

@misc{arxiv2107_07511,
  title = {{A Gentle Introduction to Conformal Prediction and Distribution-Free Uncertainty Quantification}},
  author = {Anastasios N. Angelopoulos and Stephen Bates},
  year = {2021},
  eprint = {2107.07511},
  archivePrefix = {arXiv},
  primaryClass = {cs.LG},
  url = {https://arxiv.org/abs/2107.07511v6}
}

@misc{arxiv2109_05093,
  title = {{PICARD: Parsing Incrementally for Constrained Auto-Regressive Decoding from Language Models}},
  author = {Torsten Scholak and Nathan Schucher and Dzmitry Bahdanau},
  year = {2021},
  eprint = {2109.05093},
  archivePrefix = {arXiv},
  primaryClass = {cs.CL},
  url = {https://arxiv.org/abs/2109.05093v1},
  note = {EMNLP 2021}
}

@misc{arxiv2207_05221,
  title = {{Language Models (Mostly) Know What They Know}},
  author = {Saurav Kadavath and Tom Conerly and Amanda Askell and Tom Henighan and Dawn Drain and Ethan Perez and Nicholas Schiefer and Zac Hatfield-Dodds and Nova DasSarma and Eli Tran-Johnson and Scott Johnston and Sheer El-Showk and Andy Jones and Nelson Elhage and Tristan Hume and Anna Chen and Yuntao Bai and Sam Bowman and Stanislav Fort and Deep Ganguli and Danny Hernandez and Josh Jacobson and Jackson Kernion and Shauna Kravec and Liane Lovitt and Kamal Ndousse and Catherine Olsson and Sam Ringer and Dario Amodei and Tom Brown and Jack Clark and Nicholas Joseph and Ben Mann and Sam McCandlish and Chris Olah and Jared Kaplan},
  year = {2022},
  eprint = {2207.05221},
  archivePrefix = {arXiv},
  primaryClass = {cs.CL},
  url = {https://arxiv.org/abs/2207.05221v4}
}

@misc{arxiv2209_11055,
  title = {{Efficient Few-Shot Learning Without Prompts}},
  author = {Lewis Tunstall and Nils Reimers and Unso Eun Seo Jo and Luke Bates and Daniel Korat and Moshe Wasserblat and Oren Pereg},
  year = {2022},
  eprint = {2209.11055},
  archivePrefix = {arXiv},
  primaryClass = {cs.CL},
  url = {https://arxiv.org/abs/2209.11055v1}
}

@misc{arxiv2302_09664,
  title = {{Semantic Uncertainty: Linguistic Invariances for Uncertainty Estimation in Natural Language Generation}},
  author = {Lorenz Kuhn and Yarin Gal and Sebastian Farquhar},
  year = {2023},
  eprint = {2302.09664},
  archivePrefix = {arXiv},
  primaryClass = {cs.CL},
  url = {https://arxiv.org/abs/2302.09664v3},
  note = {ICLR 2023}
}

@misc{arxiv2305_05176,
  title = {{FrugalGPT: How to Use Large Language Models While Reducing Cost and Improving Performance}},
  author = {Lingjiao Chen and Matei Zaharia and James Zou},
  year = {2023},
  eprint = {2305.05176},
  archivePrefix = {arXiv},
  primaryClass = {cs.LG},
  url = {https://arxiv.org/abs/2305.05176v1}
}

@misc{arxiv2306_05685,
  title = {{Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena}},
  author = {Lianmin Zheng and Wei-Lin Chiang and Ying Sheng and Siyuan Zhuang and Zhanghao Wu and Yonghao Zhuang and Zi Lin and Zhuohan Li and Dacheng Li and Eric P. Xing and Hao Zhang and Joseph E. Gonzalez and Ion Stoica},
  year = {2023},
  eprint = {2306.05685},
  archivePrefix = {arXiv},
  primaryClass = {cs.CL},
  url = {https://arxiv.org/abs/2306.05685v4},
  note = {NeurIPS Datasets and Benchmarks 2023}
}

@misc{arxiv2306_07536,
  title = {{TART: A plug-and-play Transformer module for task-agnostic reasoning}},
  author = {Kush Bhatia and Avanika Narayan and Christopher De Sa and Christopher Ré},
  year = {2023},
  eprint = {2306.07536},
  archivePrefix = {arXiv},
  primaryClass = {cs.LG},
  url = {https://arxiv.org/abs/2306.07536v1}
}

@misc{arxiv2307_09702,
  title = {{Efficient Guided Generation for Large Language Models}},
  author = {Brandon T. Willard and Rémi Louf},
  year = {2023},
  eprint = {2307.09702},
  archivePrefix = {arXiv},
  primaryClass = {cs.CL},
  url = {https://arxiv.org/abs/2307.09702v4}
}

@misc{arxiv2307_13565,
  title = {{Decision-Focused Learning: Foundations, State of the Art, Benchmark and Future Opportunities}},
  author = {Jayanta Mandi and James Kotary and Senne Berden and Maxime Mulamba and Victor Bucarey and Tias Guns and Ferdinando Fioretto},
  year = {2023},
  eprint = {2307.13565},
  archivePrefix = {arXiv},
  primaryClass = {cs.LG},
  url = {https://arxiv.org/abs/2307.13565v4},
  doi = {10.1613/jair.1.15320},
  note = {Journal of Artificial Intelligence Research 81 (2024) 1623-1701}
}

@misc{arxiv2309_04992,
  title = {{Mitigating Word Bias in Zero-shot Prompt-based Classifiers}},
  author = {Adian Liusie and Potsawee Manakul and Mark J. F. Gales},
  year = {2023},
  eprint = {2309.04992},
  archivePrefix = {arXiv},
  primaryClass = {cs.CL},
  url = {https://arxiv.org/abs/2309.04992v1}
}

@misc{arxiv2310_05921,
  title = {{Conformal Decision Theory: Safe Autonomous Decisions from Imperfect Predictions}},
  author = {Jordan Lekeufack and Anastasios N. Angelopoulos and Andrea Bajcsy and Michael I. Jordan and Jitendra Malik},
  year = {2023},
  eprint = {2310.05921},
  archivePrefix = {arXiv},
  primaryClass = {stat.ML},
  url = {https://arxiv.org/abs/2310.05921v3}
}

@misc{arxiv2310_08491,
  title = {{Prometheus: Inducing Fine-grained Evaluation Capability in Language Models}},
  author = {Seungone Kim and Jamin Shin and Yejin Cho and Joel Jang and Shayne Longpre and Hwaran Lee and Sangdoo Yun and Seongjin Shin and Sungdong Kim and James Thorne and Minjoon Seo},
  year = {2023},
  eprint = {2310.08491},
  archivePrefix = {arXiv},
  primaryClass = {cs.CL},
  url = {https://arxiv.org/abs/2310.08491v2},
  note = {ICLR 2024}
}

@misc{arxiv2310_11324,
  title = {{Quantifying Language Models' Sensitivity to Spurious Features in Prompt Design or: How I learned to start worrying about prompt formatting}},
  author = {Melanie Sclar and Yejin Choi and Yulia Tsvetkov and Alane Suhr},
  year = {2023},
  eprint = {2310.11324},
  archivePrefix = {arXiv},
  primaryClass = {cs.CL},
  url = {https://arxiv.org/abs/2310.11324v2},
  note = {ICLR 2024}
}

@misc{arxiv2311_08526,
  title = {{GLiNER: Generalist Model for Named Entity Recognition using Bidirectional Transformer}},
  author = {Urchade Zaratiana and Nadi Tomeh and Pierre Holat and Thierry Charnois},
  year = {2023},
  eprint = {2311.08526},
  archivePrefix = {arXiv},
  primaryClass = {cs.CL},
  url = {https://arxiv.org/abs/2311.08526v1}
}

@misc{arxiv2312_07104,
  title = {{SGLang: Efficient Execution of Structured Language Model Programs}},
  author = {Lianmin Zheng and Liangsheng Yin and Zhiqiang Xie and Chuyue Sun and Jeff Huang and Cody Hao Yu and Shiyi Cao and Christos Kozyrakis and Ion Stoica and Joseph E. Gonzalez and Clark Barrett and Ying Sheng},
  year = {2023},
  eprint = {2312.07104},
  archivePrefix = {arXiv},
  primaryClass = {cs.AI},
  url = {https://arxiv.org/abs/2312.07104v2}
}

@misc{arxiv2403_13787,
  title = {{RewardBench: Evaluating Reward Models for Language Modeling}},
  author = {Nathan Lambert and Valentina Pyatkin and Jacob Morrison and LJ Miranda and Bill Yuchen Lin and Khyathi Chandu and Nouha Dziri and Sachin Kumar and Tom Zick and Yejin Choi and Noah A. Smith and Hannaneh Hajishirzi},
  year = {2024},
  eprint = {2403.13787},
  archivePrefix = {arXiv},
  primaryClass = {cs.LG},
  url = {https://arxiv.org/abs/2403.13787v2}
}

@misc{arxiv2404_13503,
  title = {{Calibration Error for Decision Making}},
  author = {Lunjia Hu and Yifan Wu},
  year = {2024},
  eprint = {2404.13503},
  archivePrefix = {arXiv},
  primaryClass = {cs.LG},
  url = {https://arxiv.org/abs/2404.13503v5},
  note = {FOCS 2024}
}

@misc{arxiv2406_18665,
  title = {{RouteLLM: Learning to Route LLMs with Preference Data}},
  author = {Isaac Ong and Amjad Almahairi and Vincent Wu and Wei-Lin Chiang and Tianhao Wu and Joseph E. Gonzalez and M Waleed Kadous and Ion Stoica},
  year = {2024},
  eprint = {2406.18665},
  archivePrefix = {arXiv},
  primaryClass = {cs.LG},
  url = {https://arxiv.org/abs/2406.18665v4}
}

@misc{arxiv2411_15100,
  title = {{XGrammar: Flexible and Efficient Structured Generation Engine for Large Language Models}},
  author = {Yixin Dong and Charlie F. Ruan and Yaxing Cai and Ruihang Lai and Ziyi Xu and Yilong Zhao and Tianqi Chen},
  year = {2024},
  eprint = {2411.15100},
  archivePrefix = {arXiv},
  primaryClass = {cs.CL},
  url = {https://arxiv.org/abs/2411.15100v3},
  note = {MLSys 2025}
}

@misc{arxiv2412_13663,
  title = {{Smarter, Better, Faster, Longer: A Modern Bidirectional Encoder for Fast, Memory Efficient, and Long Context Finetuning and Inference}},
  author = {Benjamin Warner and Antoine Chaffin and Benjamin Clavié and Orion Weller and Oskar Hallström and Said Taghadouini and Alexis Gallagher and Raja Biswas and Faisal Ladhak and Tom Aarsen and Nathan Cooper and Griffin Adams and Jeremy Howard and Iacopo Poli},
  year = {2024},
  eprint = {2412.13663},
  archivePrefix = {arXiv},
  primaryClass = {cs.CL},
  url = {https://arxiv.org/abs/2412.13663v2}
}

@misc{arxiv2501_10868,
  title = {{JSONSchemaBench: A Rigorous Benchmark of Structured Outputs for Language Models}},
  author = {Saibo Geng and Hudson Cooper and Michał Moskal and Samuel Jenkins and Julian Berman and Nathan Ranchin and Robert West and Eric Horvitz and Harsha Nori},
  year = {2025},
  eprint = {2501.10868},
  archivePrefix = {arXiv},
  primaryClass = {cs.CL},
  url = {https://arxiv.org/abs/2501.10868v3}
}

@misc{arxiv2502_17419,
  title = {{From System 1 to System 2: A Survey of Reasoning Large Language Models}},
  author = {Zhong-Zhi Li and Duzhen Zhang and Ming-Liang Zhang and Jiaxin Zhang and Zengyan Liu and Yuxuan Yao and Haotian Xu and Junhao Zheng and Pei-Jie Wang and Xiuyi Chen and Yingying Zhang and Fei Yin and Jiahua Dong and Zhiwei Li and Bao-Long Bi and Ling-Rui Mei and Junfeng Fang and Xiao Liang and Zhijiang Guo and Le Song and Cheng-Lin Liu},
  year = {2025},
  eprint = {2502.17419},
  archivePrefix = {arXiv},
  primaryClass = {cs.AI},
  url = {https://arxiv.org/abs/2502.17419v6}
}

@misc{arxiv2503_02623,
  title = {{Rewarding Doubt: A Reinforcement Learning Approach to Calibrated Confidence Expression of Large Language Models}},
  author = {David Bani-Harouni and Chantal Pellegrini and Paul Stangel and Ege Özsoy and Kamilia Zaripova and Nassir Navab and Matthias Keicher},
  year = {2025},
  eprint = {2503.02623},
  archivePrefix = {arXiv},
  primaryClass = {cs.CL},
  url = {https://arxiv.org/abs/2503.02623v6}
}

@misc{arxiv2503_15850,
  title = {{Uncertainty Quantification and Confidence Calibration in Large Language Models: A Survey}},
  author = {Xiaoou Liu and Tiejin Chen and Longchao Da and Chacha Chen and Zhen Lin and Hua Wei},
  year = {2025},
  eprint = {2503.15850},
  archivePrefix = {arXiv},
  primaryClass = {cs.CL},
  url = {https://arxiv.org/abs/2503.15850v2}
}

@misc{arxiv2503_23303,
  title = {{SalesRLAgent: A Reinforcement Learning Approach for Real-Time Sales Conversion Prediction and Optimization}},
  author = {Nandakishor M},
  year = {2025},
  eprint = {2503.23303},
  archivePrefix = {arXiv},
  primaryClass = {cs.LG},
  url = {https://arxiv.org/abs/2503.23303v1}
}

@misc{arxiv2503_24377,
  title = {{Harnessing the Reasoning Economy: A Survey of Efficient Reasoning for Large Language Models}},
  author = {Rui Wang and Hongru Wang and Boyang Xue and Yixia Li and Jianhui Pang and Shudong Liu and Yi Chen and Jiahao Qiu and Derek Fai Wong and Guanhua Chen and Heng Ji and Kam-Fai Wong},
  year = {2025},
  eprint = {2503.24377},
  archivePrefix = {arXiv},
  primaryClass = {cs.CL},
  url = {https://arxiv.org/abs/2503.24377v3}
}

@misc{arxiv2504_15582,
  title = {{Smooth Calibration and Decision Making}},
  author = {Jason Hartline and Yifan Wu and Yunran Yang},
  year = {2025},
  eprint = {2504.15582},
  archivePrefix = {arXiv},
  primaryClass = {cs.LG},
  url = {https://arxiv.org/abs/2504.15582v1},
  note = {FORC 2025}
}

@misc{arxiv2508_07662,
  title = {{GLiClass: Generalist Lightweight Model for Sequence Classification Tasks}},
  author = {Ihor Stepanov and Mykhailo Shtopko and Dmytro Vodianytskyi and Oleksandr Lukashov and Alexander Yavorskyi and Mykyta Yaroshenko},
  year = {2025},
  eprint = {2508.07662},
  archivePrefix = {arXiv},
  primaryClass = {cs.LG},
  url = {https://arxiv.org/abs/2508.07662v1}
}

@misc{arxiv2510_01237,
  title = {{Confidence-Aware Routing for Large Language Model Reliability Enhancement: A Multi-Signal Approach to Pre-Generation Hallucination Mitigation}},
  author = {Nandakishor M},
  year = {2025},
  eprint = {2510.01237},
  archivePrefix = {arXiv},
  primaryClass = {cs.CL},
  url = {https://arxiv.org/abs/2510.01237v1}
}

@misc{arxiv2510_07750,
  title = {{Calibrating Decision Robustness via Inverse Conformal Risk Control}},
  author = {Wenbin Zhou and Shixiang Zhu},
  year = {2025},
  eprint = {2510.07750},
  archivePrefix = {arXiv},
  primaryClass = {stat.ML},
  url = {https://arxiv.org/abs/2510.07750v3}
}

@misc{arxiv2511_13699,
  title = {{Efficient Calibration for Decision Making}},
  author = {Parikshit Gopalan and Konstantinos Stavropoulos and Kunal Talwar and Pranay Tankala},
  year = {2025},
  eprint = {2511.13699},
  archivePrefix = {arXiv},
  primaryClass = {cs.LG},
  url = {https://arxiv.org/abs/2511.13699v1}
}

@misc{arxiv2601_07206,
  title = {{LLMRouterBench: A Massive Benchmark and Unified Framework for LLM Routing}},
  author = {Hao Li and Yiqun Zhang and Zhaoyan Guo and Chenxu Wang and Shengji Tang and Qiaosheng Zhang and Yang Chen and Biqing Qi and Peng Ye and Lei Bai and Zhen Wang and Shuyue Hu},
  year = {2026},
  eprint = {2601.07206},
  archivePrefix = {arXiv},
  primaryClass = {cs.AI},
  url = {https://arxiv.org/abs/2601.07206v1}
}

@misc{arxiv2604_25359,
  title = {{The Structured Output Benchmark: A Multi-Source Benchmark for Evaluating Structured Output Quality in Large Language Models}},
  author = {Abhinav Kumar Singh and Harsha Vardhan Khurdula and Yoeven D Khemlani and Vineet Agarwal},
  year = {2026},
  eprint = {2604.25359},
  archivePrefix = {arXiv},
  primaryClass = {cs.CL},
  url = {https://arxiv.org/abs/2604.25359v1}
}

@misc{arxiv2606_22807,
  title = {{KaLM-Reranker-V1: Fast but Not Late Interaction for Compressed Document Reranking}},
  author = {Xinping Zhao and Jiaxin Xu and Ziqi Dai and Xin Zhang and Huiyao Chen and Shouzheng Huang and Xianhao Xiong and Danyu Tang and Xinshuo Hu and Guohong Fu and Meishan Zhang and Baotian Hu},
  year = {2026},
  eprint = {2606.22807},
  archivePrefix = {arXiv},
  primaryClass = {cs.CL},
  url = {https://arxiv.org/abs/2606.22807v3}
}

@misc{arxiv2606_30531,
  title = {{Entity Binding Failures in Tool-Augmented Agents}},
  author = {Rahul Suresh Babu and Shashank Indukuri},
  year = {2026},
  eprint = {2606.30531},
  archivePrefix = {arXiv},
  primaryClass = {cs.AI},
  url = {https://arxiv.org/abs/2606.30531v1}
}

@misc{arxiv2607_20492,
  title = {{PhantomFill: When the Form Demands an Answer, Language Models Invent One}},
  author = {Rana Muhammad Usman},
  year = {2026},
  eprint = {2607.20492},
  archivePrefix = {arXiv},
  primaryClass = {cs.LG},
  url = {https://arxiv.org/abs/2607.20492v2}
}

@misc{arxiv2608_04355,
  title = {{The Calibration Floor: Format Repair Can Masquerade as Self-Correction at Small-to-Mid Scale}},
  author = {Mingguang Chen and Bo Qu and Licheng Wang},
  year = {2026},
  eprint = {2608.04355},
  archivePrefix = {arXiv},
  primaryClass = {cs.CL},
  url = {https://arxiv.org/abs/2608.04355v1}
}

@misc{arxiv2608_06571,
  title = {{Model Confidence Under Answer-Preserving Attacks: An Informativeness-Manipulability Frontier}},
  author = {Reza Khanmohammadi and Ivan Brugere and Simerjot Kaur and Charese H. Smiley and Kundan Thind and Mohammad M. Ghassemi},
  year = {2026},
  eprint = {2608.06571},
  archivePrefix = {arXiv},
  primaryClass = {cs.CR},
  url = {https://arxiv.org/abs/2608.06571v1}
}

@misc{arxiv2608_08254,
  title = {{Your Prompt Is Not the Only Prompt: How Much Do LLMs Weight Structured-Output Schema Descriptions?}},
  author = {Sin-Ying Lin},
  year = {2026},
  eprint = {2608.08254},
  archivePrefix = {arXiv},
  primaryClass = {cs.AI},
  url = {https://arxiv.org/abs/2608.08254v1}
}

@misc{arxiv2608_19558,
  title = {{Reliable Financial Named Entity Recognition Under Domain Shift: Confidence Estimation and Selective Prediction}},
  author = {Zihao Zheng and Baichuan Li and Junyi Yao and Jiayu Long},
  year = {2026},
  eprint = {2608.19558},
  archivePrefix = {arXiv},
  primaryClass = {cs.CL},
  url = {https://arxiv.org/abs/2608.19558v2}
}

@misc{arxiv2608_20630,
  title = {{SAGE: A Unified Algebra and Self-Adaptive Execution for AI Functions in SQL}},
  author = {Xiangqi Wang and Nhan H. Pham and Oktie Hassanzadeh and Dharmashankar Subramanian and Xiangliang Zhang},
  year = {2026},
  eprint = {2608.20630},
  archivePrefix = {arXiv},
  primaryClass = {cs.AI},
  url = {https://arxiv.org/abs/2608.20630v1}
}

@misc{arxiv2609_04445,
  title = {{Conformity Breaks Conformal Prediction}},
  author = {Yibo Hu and Hanyu Su},
  year = {2026},
  eprint = {2609.04445},
  archivePrefix = {arXiv},
  primaryClass = {cs.LG},
  url = {https://arxiv.org/abs/2609.04445v1}
}

@misc{arxiv2609_07305,
  title = {{Marginal Fidelity Does Not Establish User Simulation in Demographic Synthetic Survey Panels: Response Contracts, Support Collapse and Conditioning Failure}},
  author = {Alexander Doudkin},
  year = {2026},
  eprint = {2609.07305},
  archivePrefix = {arXiv},
  primaryClass = {cs.CL},
  url = {https://arxiv.org/abs/2609.07305v1}
}

@misc{arxiv2609_13288,
  title = {{Target-Checked Reliability Score Refinement for Video Question Answering}},
  author = {Guoxiang Ren and Rohitash Chandra},
  year = {2026},
  eprint = {2609.13288},
  archivePrefix = {arXiv},
  primaryClass = {cs.CV},
  url = {https://arxiv.org/abs/2609.13288v1}
}

@misc{arxiv2609_17977,
  title = {{When to Call an LLM: A Confidence-Gated Hybrid for Cost-Effective Emotion Recognition in Conversational AI}},
  author = {Sai Babu Udayagiri and Arjun Chouhan and Ravisekhar Kanagala and Trishala Pavagada},
  year = {2026},
  eprint = {2609.17977},
  archivePrefix = {arXiv},
  primaryClass = {cs.AI},
  url = {https://arxiv.org/abs/2609.17977v1}
}
