Benchmark Run Details

Run Summary

Model gpt-4o-mini-2024-07-18
Benchmark 0015_spell_check
Normed Score 100
Run Timestamp 2025-03-26 18:34:10

Question-Level Details

Question ID Score Evaluation Time (ms) Debug Info
0015_spell_check:attention:0 100 1095 { "response": { "incorrect": "attetion", "correct": "attention" }, "expected": { "incorrect": "attetion", "correct": "attention" } }
[+]
0015_spell_check:attention:1 100 827 { "response": { "incorrect": "attantion", "correct": "attention" }, "expected": { "incorrect": "attantion", "correct": "attention" } }
[+]
0015_spell_check:attention:2 100 1580 { "response": { "incorrect": "attetion", "correct": "attention" }, "expected": { "incorrect": "attetion", "correct": "attention" } }
[+]
0015_spell_check:attention:3 100 574 { "response": { "incorrect": "attnetion", "correct": "attention" }, "expected": { "incorrect": "attnetion", "correct": "attention" } }
[+]
0015_spell_check:attention:4 100 594 { "response": { "incorrect": "attntion", "correct": "attention" }, "expected": { "incorrect": "attntion", "correct": "attention" } }
[+]
0015_spell_check:attention:5 100 632 { "response": { "incorrect": "attenion", "correct": "attention" }, "expected": { "incorrect": "attenion", "correct": "attention" } }
[+]
0015_spell_check:attention:6 100 713 { "response": { "incorrect": "attenion", "correct": "attention" }, "expected": { "incorrect": "attenion", "correct": "attention" } }
[+]
0015_spell_check:attention:7 100 567 { "response": { "incorrect": "attetion", "correct": "attention" }, "expected": { "incorrect": "attetion", "correct": "attention" } }
[+]
0015_spell_check:attention:8 100 765 { "response": { "incorrect": "atttention", "correct": "attention" }, "expected": { "incorrect": "atttention", "correct": "attention" } }
[+]
0015_spell_check:attention:9 100 716 { "response": { "incorrect": "attenttion", "correct": "attention" }, "expected": { "incorrect": "attenttion", "correct": "attention" } }
[+]
0015_spell_check:demonstrate:0 100 769 { "response": { "incorrect": "demonstraite", "correct": "demonstrate" }, "expected": { "incorrect": "demonstraite", "correct": "demonstrate" } }
[+]
0015_spell_check:demonstrate:1 100 767 { "response": { "incorrect": "deomonstrate", "correct": "demonstrate" }, "expected": { "incorrect": "deomonstrate", "correct": "demonstrate" } }
[+]
0015_spell_check:demonstrate:2 100 743 { "response": { "incorrect": "deomonstrate", "correct": "demonstrate" }, "expected": { "incorrect": "deomonstrate", "correct": "demonstrate" } }
[+]
0015_spell_check:demonstrate:3 100 690 { "response": { "incorrect": "deomonstrate", "correct": "demonstrate" }, "expected": { "incorrect": "deomonstrate", "correct": "demonstrate" } }
[+]
0015_spell_check:demonstrate:4 100 599 { "response": { "incorrect": "demanstrate", "correct": "demonstrate" }, "expected": { "incorrect": "demanstrate", "correct": "demonstrate" } }
[+]
0015_spell_check:demonstrate:5 100 1233 { "response": { "incorrect": "deomonstrate", "correct": "demonstrate" }, "expected": { "incorrect": "deomonstrate", "correct": "demonstrate" } }
[+]
0015_spell_check:demonstrate:6 100 731 { "response": { "incorrect": "deomonstrate", "correct": "demonstrate" }, "expected": { "incorrect": "deomonstrate", "correct": "demonstrate" } }
[+]
0015_spell_check:demonstrate:7 100 811 { "response": { "incorrect": "demonstarte", "correct": "demonstrate" }, "expected": { "incorrect": "demonstarte", "correct": "demonstrate" } }
[+]
0015_spell_check:demonstrate:8 100 676 { "response": { "incorrect": "deomstrate", "correct": "demonstrate" }, "expected": { "incorrect": "deomstrate", "correct": "demonstrate" } }
[+]
0015_spell_check:demonstrate:9 100 677 { "response": { "incorrect": "deomonstrate", "correct": "demonstrate" }, "expected": { "incorrect": "deomonstrate", "correct": "demonstrate" } }
[+]
0015_spell_check:laboratory:0 100 871 { "response": { "incorrect": "labratory", "correct": "laboratory" }, "expected": { "incorrect": "labratory", "correct": "laboratory" } }
[+]
0015_spell_check:laboratory:1 100 714 { "response": { "incorrect": "labratory", "correct": "laboratory" }, "expected": { "incorrect": "labratory", "correct": "laboratory" } }
[+]
0015_spell_check:laboratory:2 100 586 { "response": { "incorrect": "labratory", "correct": "laboratory" }, "expected": { "incorrect": "labratory", "correct": "laboratory" } }
[+]
0015_spell_check:laboratory:3 100 600 { "response": { "incorrect": "labratory", "correct": "laboratory" }, "expected": { "incorrect": "labratory", "correct": "laboratory" } }
[+]
0015_spell_check:laboratory:4 100 786 { "response": { "incorrect": "labratory", "correct": "laboratory" }, "expected": { "incorrect": "labratory", "correct": "laboratory" } }
[+]
0015_spell_check:laboratory:5 100 503 { "response": { "incorrect": "labratory", "correct": "laboratory" }, "expected": { "incorrect": "labratory", "correct": "laboratory" } }
[+]
0015_spell_check:laboratory:6 100 658 { "response": { "incorrect": "labratory", "correct": "laboratory" }, "expected": { "incorrect": "labratory", "correct": "laboratory" } }
[+]
0015_spell_check:laboratory:7 100 453 { "response": { "incorrect": "labratory", "correct": "laboratory" }, "expected": { "incorrect": "labratory", "correct": "laboratory" } }
[+]
0015_spell_check:laboratory:8 100 452 { "response": { "incorrect": "labratory", "correct": "laboratory" }, "expected": { "incorrect": "labratory", "correct": "laboratory" } }
[+]
0015_spell_check:laboratory:9 100 634 { "response": { "incorrect": "laberatory", "correct": "laboratory" }, "expected": { "incorrect": "laberatory", "correct": "laboratory" } }
[+]
0015_spell_check:laughter:0 100 1247 { "response": { "incorrect": "laughterr", "correct": "laughter" }, "expected": { "incorrect": "laughterr", "correct": "laughter" } }
[+]
0015_spell_check:laughter:1 100 556 { "response": { "incorrect": "laghter", "correct": "laughter" }, "expected": { "incorrect": "laghter", "correct": "laughter" } }
[+]
0015_spell_check:laughter:2 100 614 { "response": { "incorrect": "laghter", "correct": "laughter" }, "expected": { "incorrect": "laghter", "correct": "laughter" } }
[+]
0015_spell_check:laughter:3 100 908 { "response": { "incorrect": "laghter", "correct": "laughter" }, "expected": { "incorrect": "laghter", "correct": "laughter" } }
[+]
0015_spell_check:laughter:4 100 523 { "response": { "incorrect": "lafter", "correct": "laughter" }, "expected": { "incorrect": "lafter", "correct": "laughter" } }
[+]
0015_spell_check:laughter:5 100 504 { "response": { "incorrect": "laghter", "correct": "laughter" }, "expected": { "incorrect": "laghter", "correct": "laughter" } }
[+]
0015_spell_check:laughter:6 100 1135 { "response": { "incorrect": "laugther", "correct": "laughter" }, "expected": { "incorrect": "laugther", "correct": "laughter" } }
[+]
0015_spell_check:laughter:7 100 1022 { "response": { "incorrect": "laughtter", "correct": "laughter" }, "expected": { "incorrect": "laughtter", "correct": "laughter" } }
[+]
0015_spell_check:laughter:8 100 700 { "response": { "incorrect": "laughtur", "correct": "laughter" }, "expected": { "incorrect": "laughtur", "correct": "laughter" } }
[+]
0015_spell_check:laughter:9 100 578 { "response": { "incorrect": "laugther", "correct": "laughter" }, "expected": { "incorrect": "laugther", "correct": "laughter" } }
[+]
0015_spell_check:liaison:0 100 768 { "response": { "incorrect": "liason", "correct": "liaison" }, "expected": { "incorrect": "liason", "correct": "liaison" } }
[+]
0015_spell_check:liaison:1 100 541 { "response": { "incorrect": "liason", "correct": "liaison" }, "expected": { "incorrect": "liason", "correct": "liaison" } }
[+]
0015_spell_check:liaison:2 100 585 { "response": { "incorrect": "liasion", "correct": "liaison" }, "expected": { "incorrect": "liasion", "correct": "liaison" } }
[+]
0015_spell_check:liaison:3 100 595 { "response": { "incorrect": "liason", "correct": "liaison" }, "expected": { "incorrect": "liason", "correct": "liaison" } }
[+]
0015_spell_check:liaison:4 100 2476 { "response": { "incorrect": "liason", "correct": "liaison" }, "expected": { "incorrect": "liason", "correct": "liaison" } }
[+]
0015_spell_check:liaison:5 100 919 { "response": { "incorrect": "liason", "correct": "liaison" }, "expected": { "incorrect": "liason", "correct": "liaison" } }
[+]
0015_spell_check:liaison:6 100 488 { "response": { "incorrect": "liasion", "correct": "liaison" }, "expected": { "incorrect": "liasion", "correct": "liaison" } }
[+]
0015_spell_check:liaison:7 100 478 { "response": { "incorrect": "leason", "correct": "liaison" }, "expected": { "incorrect": "leason", "correct": "liaison" } }
[+]
0015_spell_check:liaison:8 100 628 { "response": { "incorrect": "liasion", "correct": "liaison" }, "expected": { "incorrect": "liasion", "correct": "liaison" } }
[+]
0015_spell_check:liaison:9 100 658 { "response": { "incorrect": "liason", "correct": "liaison" }, "expected": { "incorrect": "liason", "correct": "liaison" } }
[+]
0015_spell_check:orange:0 100 420 { "response": { "incorrect": "orrange", "correct": "orange" }, "expected": { "incorrect": "orrange", "correct": "orange" } }
[+]
0015_spell_check:orange:1 100 807 { "response": { "incorrect": "oranje", "correct": "orange" }, "expected": { "incorrect": "oranje", "correct": "orange" } }
[+]
0015_spell_check:orange:2 100 429 { "response": { "incorrect": "orang", "correct": "orange" }, "expected": { "incorrect": "orang", "correct": "orange" } }
[+]
0015_spell_check:orange:3 100 492 { "response": { "incorrect": "oranage", "correct": "orange" }, "expected": { "incorrect": "oranage", "correct": "orange" } }
[+]
0015_spell_check:orange:4 100 510 { "response": { "incorrect": "oranage", "correct": "orange" }, "expected": { "incorrect": "oranage", "correct": "orange" } }
[+]
0015_spell_check:orange:5 100 604 { "response": { "incorrect": "orrange", "correct": "orange" }, "expected": { "incorrect": "orrange", "correct": "orange" } }
[+]
0015_spell_check:orange:6 100 489 { "response": { "incorrect": "orrange", "correct": "orange" }, "expected": { "incorrect": "orrange", "correct": "orange" } }
[+]
0015_spell_check:orange:7 100 1961 { "response": { "incorrect": "oringe", "correct": "orange" }, "expected": { "incorrect": "oringe", "correct": "orange" } }
[+]
0015_spell_check:orange:8 100 490 { "response": { "incorrect": "orennge", "correct": "orange" }, "expected": { "incorrect": "orennge", "correct": "orange" } }
[+]
0015_spell_check:orange:9 100 653 { "response": { "incorrect": "orrange", "correct": "orange" }, "expected": { "incorrect": "orrange", "correct": "orange" } }
[+]
0015_spell_check:partition:0 100 516 { "response": { "incorrect": "partioned", "correct": "partitioned" }, "expected": { "incorrect": "partioned", "correct": "partitioned" } }
[+]
0015_spell_check:partition:1 100 446 { "response": { "incorrect": "partion", "correct": "partition" }, "expected": { "incorrect": "partion", "correct": "partition" } }
[+]
0015_spell_check:partition:2 100 703 { "response": { "incorrect": "partion", "correct": "partition" }, "expected": { "incorrect": "partion", "correct": "partition" } }
[+]
0015_spell_check:partition:3 100 462 { "response": { "incorrect": "partionned", "correct": "partitioned" }, "expected": { "incorrect": "partionned", "correct": "partitioned" } }
[+]
0015_spell_check:partition:4 100 599 { "response": { "incorrect": "partion", "correct": "partition" }, "expected": { "incorrect": "partion", "correct": "partition" } }
[+]
0015_spell_check:partition:5 100 649 { "response": { "incorrect": "partion", "correct": "partition" }, "expected": { "incorrect": "partion", "correct": "partition" } }
[+]
0015_spell_check:partition:6 100 922 { "response": { "incorrect": "partion", "correct": "partition" }, "expected": { "incorrect": "partion", "correct": "partition" } }
[+]
0015_spell_check:partition:7 100 512 { "response": { "incorrect": "particion", "correct": "partition" }, "expected": { "incorrect": "particion", "correct": "partition" } }
[+]
0015_spell_check:partition:8 100 406 { "response": { "incorrect": "partion", "correct": "partition" }, "expected": { "incorrect": "partion", "correct": "partition" } }
[+]
0015_spell_check:partition:9 100 922 { "response": { "incorrect": "partion", "correct": "partition" }, "expected": { "incorrect": "partion", "correct": "partition" } }
[+]
0015_spell_check:party:0 100 516 { "response": { "incorrect": "partee", "correct": "party" }, "expected": { "incorrect": "partee", "correct": "party" } }
[+]
0015_spell_check:party:1 100 614 { "response": { "incorrect": "partee", "correct": "party" }, "expected": { "incorrect": "partee", "correct": "party" } }
[+]
0015_spell_check:party:2 100 485 { "response": { "incorrect": "pary", "correct": "party" }, "expected": { "incorrect": "pary", "correct": "party" } }
[+]
0015_spell_check:party:3 100 596 { "response": { "incorrect": "partee", "correct": "party" }, "expected": { "incorrect": "partee", "correct": "party" } }
[+]
0015_spell_check:party:4 100 654 { "response": { "incorrect": "partee", "correct": "party" }, "expected": { "incorrect": "partee", "correct": "party" } }
[+]
0015_spell_check:party:5 100 512 { "response": { "incorrect": "partee", "correct": "party" }, "expected": { "incorrect": "partee", "correct": "party" } }
[+]
0015_spell_check:party:6 100 511 { "response": { "incorrect": "partee", "correct": "party" }, "expected": { "incorrect": "partee", "correct": "party" } }
[+]
0015_spell_check:party:7 100 596 { "response": { "incorrect": "partee", "correct": "party" }, "expected": { "incorrect": "partee", "correct": "party" } }
[+]
0015_spell_check:party:8 100 838 { "response": { "incorrect": "partee", "correct": "party" }, "expected": { "incorrect": "partee", "correct": "party" } }
[+]
0015_spell_check:party:9 100 555 { "response": { "incorrect": "pary", "correct": "party" }, "expected": { "incorrect": "pary", "correct": "party" } }
[+]
0015_spell_check:stable:0 100 500 { "response": { "incorrect": "staable", "correct": "stable" }, "expected": { "incorrect": "staable", "correct": "stable" } }
[+]
0015_spell_check:stable:1 100 761 { "response": { "incorrect": "stabe", "correct": "stable" }, "expected": { "incorrect": "stabe", "correct": "stable" } }
[+]
0015_spell_check:stable:2 100 640 { "response": { "incorrect": "stabal", "correct": "stable" }, "expected": { "incorrect": "stabal", "correct": "stable" } }
[+]
0015_spell_check:stable:3 100 615 { "response": { "incorrect": "staible", "correct": "stable" }, "expected": { "incorrect": "staible", "correct": "stable" } }
[+]
0015_spell_check:stable:4 100 610 { "response": { "incorrect": "stabel", "correct": "stable" }, "expected": { "incorrect": "stabel", "correct": "stable" } }
[+]
0015_spell_check:stable:5 100 614 { "response": { "incorrect": "staible", "correct": "stable" }, "expected": { "incorrect": "staible", "correct": "stable" } }
[+]
0015_spell_check:stable:6 100 456 { "response": { "incorrect": "stabl", "correct": "stable" }, "expected": { "incorrect": "stabl", "correct": "stable" } }
[+]
0015_spell_check:stable:7 100 568 { "response": { "incorrect": "stabe", "correct": "stable" }, "expected": { "incorrect": "stabe", "correct": "stable" } }
[+]
0015_spell_check:stable:8 100 615 { "response": { "incorrect": "stabble", "correct": "stable" }, "expected": { "incorrect": "stabble", "correct": "stable" } }
[+]
0015_spell_check:stable:9 100 610 { "response": { "incorrect": "stayble", "correct": "stable" }, "expected": { "incorrect": "stayble", "correct": "stable" } }
[+]
0015_spell_check:table:0 100 544 { "response": { "incorrect": "tabele", "correct": "table" }, "expected": { "incorrect": "tabele", "correct": "table" } }
[+]
0015_spell_check:table:1 100 608 { "response": { "incorrect": "tabble", "correct": "table" }, "expected": { "incorrect": "tabble", "correct": "table" } }
[+]
0015_spell_check:table:2 100 590 { "response": { "incorrect": "tabel", "correct": "table" }, "expected": { "incorrect": "tabel", "correct": "table" } }
[+]
0015_spell_check:table:3 100 613 { "response": { "incorrect": "tabble", "correct": "table" }, "expected": { "incorrect": "tabble", "correct": "table" } }
[+]
0015_spell_check:table:4 100 499 { "response": { "incorrect": "tabble", "correct": "table" }, "expected": { "incorrect": "tabble", "correct": "table" } }
[+]
0015_spell_check:table:5 100 576 { "response": { "incorrect": "tabl", "correct": "table" }, "expected": { "incorrect": "tabl", "correct": "table" } }
[+]
0015_spell_check:table:6 100 561 { "response": { "incorrect": "tabl", "correct": "table" }, "expected": { "incorrect": "tabl", "correct": "table" } }
[+]
0015_spell_check:table:7 100 717 { "response": { "incorrect": "tabel", "correct": "table" }, "expected": { "incorrect": "tabel", "correct": "table" } }
[+]
0015_spell_check:table:8 100 614 { "response": { "incorrect": "tabel", "correct": "table" }, "expected": { "incorrect": "tabel", "correct": "table" } }
[+]
0015_spell_check:table:9 100 613 { "response": { "incorrect": "tabel", "correct": "table" }, "expected": { "incorrect": "tabel", "correct": "table" } }
[+]