model: gpt-4.1-mini-2025-04-14
judge_model: gpt-4o-mini-2024-07-18
inference_model: gpt-4o-mini-2024-07-18
mode: few_shot
strategy: none
n_hypotheses: 10
n_samples: 20
n_test_instances: 100
n_tasks: 7
avg_clarity: 3.4 +/- 0.44
avg_novelty: 2.8 +/- 0.265
avg_plausibility: 4.022 +/- 0.396
avg_quality: 3.407 +/- 0.281
avg_diversity: 0.461 +/- 0.049… See the full description on the dataset page:
https://huggingface.co/datasets/strategy-scope/hypobench-few_shot-gpt41mini-20260413_143226.