model: gpt-5.4-2026-03-05
judge_model: gpt-4o-mini-2024-07-18
inference_model: gpt-4o-mini-2024-07-18
mode: zero_shot
strategy: none
n_hypotheses: 10
n_samples: 20
n_test_instances: 100
n_tasks: 7
avg_clarity: 3.398 +/- 0.367
avg_novelty: 2.845 +/- 0.728
avg_plausibility: 3.768 +/- 0.161
avg_quality: 3.337 +/- 0.299
avg_diversity: 0.359 +/- 0.092… See the full description on the dataset page:
https://huggingface.co/datasets/strategy-scope/hypobench-zero_shot-gpt5420260305-20260415_002452.