PlanBench: An Extensible Benchmark for Evaluating Large Language Models on Planning and Reasoning about Change
| |
|
|
apice = {PlanbenchNeurips2023},
author = {Karthik Valmeekam and Matthew Marquez and Alberto Olmo and Sarath Sreedharan and Subbarao Kambhampati},
booktitle = {37th Conference on Neural Information Processing Systems (NeurIPS 2023) -- Datasets and Benchmarks Track},
doi = {10.52202/075280-1693},
title = {PlanBench: An Extensible Benchmark for Evaluating Large Language Models on Planning and Reasoning about Change},
url = {https://proceedings.neurips.cc/paper_files/paper/2023/hash/7a92bcdede88c7afd108072faf5485c8-Abstract-Datasets_and_Benchmarks.html},
year = 2023
}