1352 lines
57 KiB
JSON
1352 lines
57 KiB
JSON
{
|
||
"generated_at": "2026-06-22T20:27:49Z",
|
||
"ckpt": "runs/v10_348m_cone/ckpt_0210000.pt",
|
||
"step": 210000,
|
||
"device": "cuda",
|
||
"threads": 8,
|
||
"diagnostics": {
|
||
"transitive": {
|
||
"depth_1": {
|
||
"N": 128,
|
||
"chance": 0.5,
|
||
"first_name_accuracy": 0.3203125,
|
||
"answer_anywhere_rate": 0.3203125,
|
||
"ci95_first_name": [
|
||
0.24572896282451612,
|
||
0.40536750445136643
|
||
],
|
||
"by_type": {
|
||
"tallest": {
|
||
"N": 64,
|
||
"first_name_accuracy": 0.5,
|
||
"answer_anywhere_rate": 0.5,
|
||
"ci95_first_name": [
|
||
0.38101888493214947,
|
||
0.6189811150678505
|
||
]
|
||
},
|
||
"shortest": {
|
||
"N": 64,
|
||
"first_name_accuracy": 0.140625,
|
||
"answer_anywhere_rate": 0.140625,
|
||
"ci95_first_name": [
|
||
0.07578487006414436,
|
||
0.24616508394637426
|
||
]
|
||
}
|
||
},
|
||
"examples": [
|
||
{
|
||
"type": "tallest",
|
||
"prompt": "Finn is taller than Cara. Of all of them, the tallest is",
|
||
"gold": "Finn",
|
||
"gen": " Dane.\n\nVerification: Consider a different set: {1,",
|
||
"first_name": "",
|
||
"first_name_ok": false,
|
||
"answer_anywhere": false
|
||
},
|
||
{
|
||
"type": "shortest",
|
||
"prompt": "Finn is taller than Cara. Of all of them, the shortest is",
|
||
"gold": "Cara",
|
||
"gen": " the one that is shorter than the tallest.\n\nNow, let's say",
|
||
"first_name": "",
|
||
"first_name_ok": false,
|
||
"answer_anywhere": false
|
||
},
|
||
{
|
||
"type": "tallest",
|
||
"prompt": "Hugo is taller than Ben. Of all of them, the tallest is",
|
||
"gold": "Hugo",
|
||
"gen": " Ben.\n\nLet's double-check: We can verify this by checking",
|
||
"first_name": "Ben",
|
||
"first_name_ok": false,
|
||
"answer_anywhere": false
|
||
},
|
||
{
|
||
"type": "shortest",
|
||
"prompt": "Hugo is taller than Ben. Of all of them, the shortest is",
|
||
"gold": "Ben",
|
||
"gen": " Ben, the tallest is Ben.\n\nLet's double-check: We",
|
||
"first_name": "Ben",
|
||
"first_name_ok": true,
|
||
"answer_anywhere": true
|
||
},
|
||
{
|
||
"type": "tallest",
|
||
"prompt": "Gina is taller than Anna. Of all of them, the tallest is",
|
||
"gold": "Gina",
|
||
"gen": " Anna.\n\nLet's double-check: We can verify this by grouping",
|
||
"first_name": "Anna",
|
||
"first_name_ok": false,
|
||
"answer_anywhere": false
|
||
},
|
||
{
|
||
"type": "shortest",
|
||
"prompt": "Gina is taller than Anna. Of all of them, the shortest is",
|
||
"gold": "Anna",
|
||
"gen": " Alice, the tallest.\n\nLet's double-check: We can verify",
|
||
"first_name": "",
|
||
"first_name_ok": false,
|
||
"answer_anywhere": false
|
||
},
|
||
{
|
||
"type": "tallest",
|
||
"prompt": "Ben is taller than Evan. Of all of them, the tallest is",
|
||
"gold": "Ben",
|
||
"gen": " Ben, and the tallest is Evan.\n\nLet's double-check",
|
||
"first_name": "Ben",
|
||
"first_name_ok": true,
|
||
"answer_anywhere": true
|
||
},
|
||
{
|
||
"type": "shortest",
|
||
"prompt": "Ben is taller than Evan. Of all of them, the shortest is",
|
||
"gold": "Evan",
|
||
"gen": " Ben, the tallest.\n\nLet's double-check: We can verify",
|
||
"first_name": "Ben",
|
||
"first_name_ok": false,
|
||
"answer_anywhere": false
|
||
},
|
||
{
|
||
"type": "tallest",
|
||
"prompt": "Cara is taller than Ben. Of all of them, the tallest is",
|
||
"gold": "Cara",
|
||
"gen": " Cara.\n\nLet's double-check: We can verify this by",
|
||
"first_name": "Cara",
|
||
"first_name_ok": true,
|
||
"answer_anywhere": true
|
||
},
|
||
{
|
||
"type": "shortest",
|
||
"prompt": "Cara is taller than Ben. Of all of them, the shortest is",
|
||
"gold": "Ben",
|
||
"gen": " Ben, the tallest.\n\nLet's double-check: We can verify",
|
||
"first_name": "Ben",
|
||
"first_name_ok": true,
|
||
"answer_anywhere": true
|
||
},
|
||
{
|
||
"type": "tallest",
|
||
"prompt": "Gina is taller than Cara. Of all of them, the tallest is",
|
||
"gold": "Gina",
|
||
"gen": " Cara.\n\nLet's double-check: We can verify this by",
|
||
"first_name": "Cara",
|
||
"first_name_ok": false,
|
||
"answer_anywhere": false
|
||
},
|
||
{
|
||
"type": "shortest",
|
||
"prompt": "Gina is taller than Cara. Of all of them, the shortest is",
|
||
"gold": "Cara",
|
||
"gen": " the tallest.\n\nLet's double-check: We can verify this by",
|
||
"first_name": "",
|
||
"first_name_ok": false,
|
||
"answer_anywhere": false
|
||
}
|
||
]
|
||
},
|
||
"depth_2": {
|
||
"N": 128,
|
||
"chance": 0.3333333333333333,
|
||
"first_name_accuracy": 0.2421875,
|
||
"answer_anywhere_rate": 0.328125,
|
||
"ci95_first_name": [
|
||
0.17618632286113767,
|
||
0.3232129562738243
|
||
],
|
||
"by_type": {
|
||
"tallest": {
|
||
"N": 64,
|
||
"first_name_accuracy": 0.34375,
|
||
"answer_anywhere_rate": 0.34375,
|
||
"ci95_first_name": [
|
||
0.2392297503546903,
|
||
0.46596588182379606
|
||
]
|
||
},
|
||
"shortest": {
|
||
"N": 64,
|
||
"first_name_accuracy": 0.140625,
|
||
"answer_anywhere_rate": 0.3125,
|
||
"ci95_first_name": [
|
||
0.07578487006414436,
|
||
0.24616508394637426
|
||
]
|
||
}
|
||
},
|
||
"examples": [
|
||
{
|
||
"type": "tallest",
|
||
"prompt": "Evan is taller than Finn. Finn is taller than Dana. Of all of them, the tallest is",
|
||
"gold": "Evan",
|
||
"gen": " Evan.\n\nVerification: Consider a different set: {Alice,",
|
||
"first_name": "Evan",
|
||
"first_name_ok": true,
|
||
"answer_anywhere": true
|
||
},
|
||
{
|
||
"type": "shortest",
|
||
"prompt": "Evan is taller than Finn. Finn is taller than Dana. Of all of them, the shortest is",
|
||
"gold": "Dana",
|
||
"gen": " Evan, the tallest.\n\nVerification: Consider a different set:",
|
||
"first_name": "Evan",
|
||
"first_name_ok": false,
|
||
"answer_anywhere": false
|
||
},
|
||
{
|
||
"type": "tallest",
|
||
"prompt": "Cara is taller than Finn. Finn is taller than Anna. Of all of them, the tallest is",
|
||
"gold": "Cara",
|
||
"gen": " Cara.\n\nVerification: Consider a different set: {Alice,",
|
||
"first_name": "Cara",
|
||
"first_name_ok": true,
|
||
"answer_anywhere": true
|
||
},
|
||
{
|
||
"type": "shortest",
|
||
"prompt": "Cara is taller than Finn. Finn is taller than Anna. Of all of them, the shortest is",
|
||
"gold": "Anna",
|
||
"gen": " Finn, the tallest.\n\nVerification: Consider a different set: {",
|
||
"first_name": "Finn",
|
||
"first_name_ok": false,
|
||
"answer_anywhere": false
|
||
},
|
||
{
|
||
"type": "tallest",
|
||
"prompt": "Gina is taller than Evan. Evan is taller than Hugo. Of all of them, the tallest is",
|
||
"gold": "Gina",
|
||
"gen": " Evan.\n\nLet's double-check: We can verify this by",
|
||
"first_name": "Evan",
|
||
"first_name_ok": false,
|
||
"answer_anywhere": false
|
||
},
|
||
{
|
||
"type": "shortest",
|
||
"prompt": "Gina is taller than Evan. Evan is taller than Hugo. Of all of them, the shortest is",
|
||
"gold": "Hugo",
|
||
"gen": " Evan, the tallest.\n\nLet's double-check: We can",
|
||
"first_name": "Evan",
|
||
"first_name_ok": false,
|
||
"answer_anywhere": false
|
||
},
|
||
{
|
||
"type": "tallest",
|
||
"prompt": "Evan is taller than Gina. Gina is taller than Anna. Of all of them, the tallest is",
|
||
"gold": "Evan",
|
||
"gen": " Anna.\n\nVerification: Consider a different set: {1, 2",
|
||
"first_name": "Anna",
|
||
"first_name_ok": false,
|
||
"answer_anywhere": false
|
||
},
|
||
{
|
||
"type": "shortest",
|
||
"prompt": "Evan is taller than Gina. Gina is taller than Anna. Of all of them, the shortest is",
|
||
"gold": "Anna",
|
||
"gen": " Anna.\n\nVerification: Consider a different set: {1, 2",
|
||
"first_name": "Anna",
|
||
"first_name_ok": true,
|
||
"answer_anywhere": true
|
||
},
|
||
{
|
||
"type": "tallest",
|
||
"prompt": "Gina is taller than Anna. Anna is taller than Cara. Of all of them, the tallest is",
|
||
"gold": "Gina",
|
||
"gen": " Cara.\n\nVerification: Consider a different set: {1,",
|
||
"first_name": "Cara",
|
||
"first_name_ok": false,
|
||
"answer_anywhere": false
|
||
},
|
||
{
|
||
"type": "shortest",
|
||
"prompt": "Gina is taller than Anna. Anna is taller than Cara. Of all of them, the shortest is",
|
||
"gold": "Cara",
|
||
"gen": " Anna.\n\nVerification: Consider a different set: {1, 2",
|
||
"first_name": "Anna",
|
||
"first_name_ok": false,
|
||
"answer_anywhere": false
|
||
},
|
||
{
|
||
"type": "tallest",
|
||
"prompt": "Gina is taller than Dana. Dana is taller than Ben. Of all of them, the tallest is",
|
||
"gold": "Gina",
|
||
"gen": " Anna.\n\nLet's double-check: We can verify this by listing",
|
||
"first_name": "Anna",
|
||
"first_name_ok": false,
|
||
"answer_anywhere": false
|
||
},
|
||
{
|
||
"type": "shortest",
|
||
"prompt": "Gina is taller than Dana. Dana is taller than Ben. Of all of them, the shortest is",
|
||
"gold": "Ben",
|
||
"gen": " Alice, the tallest.\n\nVerification: Consider a different set: {",
|
||
"first_name": "",
|
||
"first_name_ok": false,
|
||
"answer_anywhere": false
|
||
}
|
||
]
|
||
},
|
||
"depth_3": {
|
||
"N": 128,
|
||
"chance": 0.25,
|
||
"first_name_accuracy": 0.0546875,
|
||
"answer_anywhere_rate": 0.0625,
|
||
"ci95_first_name": [
|
||
0.02673986088319332,
|
||
0.10858616671355914
|
||
],
|
||
"by_type": {
|
||
"tallest": {
|
||
"N": 64,
|
||
"first_name_accuracy": 0.078125,
|
||
"answer_anywhere_rate": 0.078125,
|
||
"ci95_first_name": [
|
||
0.03383067809543547,
|
||
0.1701975287864777
|
||
]
|
||
},
|
||
"shortest": {
|
||
"N": 64,
|
||
"first_name_accuracy": 0.03125,
|
||
"answer_anywhere_rate": 0.046875,
|
||
"ci95_first_name": [
|
||
0.00861195771251659,
|
||
0.10697493882294247
|
||
]
|
||
}
|
||
},
|
||
"examples": [
|
||
{
|
||
"type": "tallest",
|
||
"prompt": "Gina is taller than Anna. Anna is taller than Evan. Evan is taller than Ben. Of all of them, the tallest is",
|
||
"gold": "Gina",
|
||
"gen": " Anna.\n\nLet's double-check: We can verify this by listing",
|
||
"first_name": "Anna",
|
||
"first_name_ok": false,
|
||
"answer_anywhere": false
|
||
},
|
||
{
|
||
"type": "shortest",
|
||
"prompt": "Gina is taller than Anna. Anna is taller than Evan. Evan is taller than Ben. Of all of them, the shortest is",
|
||
"gold": "Ben",
|
||
"gen": " Anna, the tallest is Anna, and the tallest is Evan.\n\n",
|
||
"first_name": "Anna",
|
||
"first_name_ok": false,
|
||
"answer_anywhere": false
|
||
},
|
||
{
|
||
"type": "tallest",
|
||
"prompt": "Cara is taller than Hugo. Hugo is taller than Gina. Gina is taller than Dana. Of all of them, the tallest is",
|
||
"gold": "Cara",
|
||
"gen": " g(Dana).\n\nReasoning: We are given three statements",
|
||
"first_name": "Dana",
|
||
"first_name_ok": false,
|
||
"answer_anywhere": false
|
||
},
|
||
{
|
||
"type": "shortest",
|
||
"prompt": "Cara is taller than Hugo. Hugo is taller than Gina. Gina is taller than Dana. Of all of them, the shortest is",
|
||
"gold": "Dana",
|
||
"gen": " g, the tallest is h, the shortest is d, the tallest is c",
|
||
"first_name": "",
|
||
"first_name_ok": false,
|
||
"answer_anywhere": false
|
||
},
|
||
{
|
||
"type": "tallest",
|
||
"prompt": "Cara is taller than Evan. Evan is taller than Ben. Ben is taller than Anna. Of all of them, the tallest is",
|
||
"gold": "Cara",
|
||
"gen": " Evan.\n\nVerification: Consider a different set: {Alice,",
|
||
"first_name": "Evan",
|
||
"first_name_ok": false,
|
||
"answer_anywhere": false
|
||
},
|
||
{
|
||
"type": "shortest",
|
||
"prompt": "Cara is taller than Evan. Evan is taller than Ben. Ben is taller than Anna. Of all of them, the shortest is",
|
||
"gold": "Anna",
|
||
"gen": " Ben.\n\nVerification: Consider a different set: {Alice, Bob",
|
||
"first_name": "Ben",
|
||
"first_name_ok": false,
|
||
"answer_anywhere": false
|
||
},
|
||
{
|
||
"type": "tallest",
|
||
"prompt": "Cara is taller than Gina. Gina is taller than Dana. Dana is taller than Anna. Of all of them, the tallest is",
|
||
"gold": "Cara",
|
||
"gen": " Anna.\n\nVerification: Consider a different set: {1, 2",
|
||
"first_name": "Anna",
|
||
"first_name_ok": false,
|
||
"answer_anywhere": false
|
||
},
|
||
{
|
||
"type": "shortest",
|
||
"prompt": "Cara is taller than Gina. Gina is taller than Dana. Dana is taller than Anna. Of all of them, the shortest is",
|
||
"gold": "Anna",
|
||
"gen": " Dana.\n\nVerification: Consider a different set: {1, 2",
|
||
"first_name": "Dana",
|
||
"first_name_ok": false,
|
||
"answer_anywhere": false
|
||
},
|
||
{
|
||
"type": "tallest",
|
||
"prompt": "Anna is taller than Cara. Cara is taller than Gina. Gina is taller than Ben. Of all of them, the tallest is",
|
||
"gold": "Anna",
|
||
"gen": " Cara.\n\nLet's double-check: We can verify this by",
|
||
"first_name": "Cara",
|
||
"first_name_ok": false,
|
||
"answer_anywhere": false
|
||
},
|
||
{
|
||
"type": "shortest",
|
||
"prompt": "Anna is taller than Cara. Cara is taller than Gina. Gina is taller than Ben. Of all of them, the shortest is",
|
||
"gold": "Ben",
|
||
"gen": " Cara. So the shortest height is Cara.\n\nLet's double",
|
||
"first_name": "Cara",
|
||
"first_name_ok": false,
|
||
"answer_anywhere": false
|
||
},
|
||
{
|
||
"type": "tallest",
|
||
"prompt": "Cara is taller than Ben. Ben is taller than Dana. Dana is taller than Anna. Of all of them, the tallest is",
|
||
"gold": "Cara",
|
||
"gen": " Anna.\n\nVerification: Consider a different set: {Alice, Bob",
|
||
"first_name": "Anna",
|
||
"first_name_ok": false,
|
||
"answer_anywhere": false
|
||
},
|
||
{
|
||
"type": "shortest",
|
||
"prompt": "Cara is taller than Ben. Ben is taller than Dana. Dana is taller than Anna. Of all of them, the shortest is",
|
||
"gold": "Anna",
|
||
"gen": " Ben. So the shortest height is Ben.\n\nLet's double-check",
|
||
"first_name": "Ben",
|
||
"first_name_ok": false,
|
||
"answer_anywhere": false
|
||
}
|
||
]
|
||
}
|
||
},
|
||
"math_extraction": {
|
||
"N": 256,
|
||
"by_source": {
|
||
"corpus_v3/L2/math-drill-v1/canonical.jsonl": {
|
||
"N": 64,
|
||
"first_numeric_accuracy": 0.21875,
|
||
"final_numeric_accuracy": 0.203125,
|
||
"answer_anywhere_rate": 0.375,
|
||
"first_line_answer_anywhere_rate": 0.375,
|
||
"ci95_answer_anywhere": [
|
||
0.2666580388810532,
|
||
0.4974984668617358
|
||
],
|
||
"examples": [
|
||
{
|
||
"problem": "Add 7 and 2.",
|
||
"gold": "9",
|
||
"gen": " The sum is 14. The failure mode to avoid is assuming the answer must be positive without verifying the equation's structure.\nConfidence: 0.85",
|
||
"first_line": "The sum is 14. The failure mode to avoid is assuming the answer must be positive without verifying the equation's structure.",
|
||
"first_numeric_match": false,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": false,
|
||
"first_line_answer_anywhere": false
|
||
},
|
||
{
|
||
"problem": "83 + 49 =",
|
||
"gold": "132",
|
||
"gen": " 50\n\nSo, the sum of the first and last numbers in the sequence is 50.\n\nExample 2: Find the sum of the first five terms of the sequence:\n\n1, 3, 5, 7, 9\n",
|
||
"first_line": "50",
|
||
"first_numeric_match": false,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": false,
|
||
"first_line_answer_anywhere": false
|
||
},
|
||
{
|
||
"problem": "What's 23 plus 91?",
|
||
"gold": "114",
|
||
"gen": " 23 + 91 = 124\n\nSo, the sum of the first and last numbers in our sequence is 124.\n\nNow, let's try another example. Suppose we have the sequence: 1, 2, 3, 4, 5",
|
||
"first_line": "23 + 91 = 124",
|
||
"first_numeric_match": false,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": false,
|
||
"first_line_answer_anywhere": false
|
||
},
|
||
{
|
||
"problem": "8 minus 1 equals what?",
|
||
"gold": "7",
|
||
"gen": " 5\n\nSo, the answer is 5.\nI'm confident in this answer.",
|
||
"first_line": "5",
|
||
"first_numeric_match": false,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": false,
|
||
"first_line_answer_anywhere": false
|
||
},
|
||
{
|
||
"problem": "Divide: 65 ÷ 11",
|
||
"gold": "5 remainder 10",
|
||
"gen": " 65 × 11 = 545\n\nLet's double-check: Verification by reverse division: if 65 × 11 = 544, then 544 ÷ 11 should equal 65. Indeed 544 ÷ 11 = 65. This confirms 65 × 11 = 544.",
|
||
"first_line": "65 × 11 = 545",
|
||
"first_numeric_match": false,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": false,
|
||
"first_line_answer_anywhere": false
|
||
},
|
||
{
|
||
"problem": "? × 3 = 18",
|
||
"gold": "6",
|
||
"gen": " 18\n\nSo, the product of the first and second numbers is 18.\n\nLet's double-check: We can verify this by using the distributive property: The product of the first and third numbers is 12 × 3 = 36",
|
||
"first_line": "18",
|
||
"first_numeric_match": false,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": false,
|
||
"first_line_answer_anywhere": false
|
||
}
|
||
]
|
||
},
|
||
"corpus_v3/L2/math-k6-v2/canonical.jsonl": {
|
||
"N": 64,
|
||
"first_numeric_accuracy": 0.125,
|
||
"final_numeric_accuracy": 0.25,
|
||
"answer_anywhere_rate": 0.484375,
|
||
"first_line_answer_anywhere_rate": 0.4375,
|
||
"ci95_answer_anywhere": [
|
||
0.36633348565783896,
|
||
0.6041860775600096
|
||
],
|
||
"examples": [
|
||
{
|
||
"problem": "Draw a clock showing 3:00. What time is it? Write the time in numbers.",
|
||
"gold": "3:00",
|
||
"gen": " The time is 3:00. The answer is 3:00.\nI'm confident in this answer.",
|
||
"first_line": "The time is 3:00. The answer is 3:00.",
|
||
"first_numeric_match": false,
|
||
"final_numeric_match": true,
|
||
"answer_anywhere": true,
|
||
"first_line_answer_anywhere": true
|
||
},
|
||
{
|
||
"problem": "Mia has 8 stickers. She gets 5 more stickers. How many stickers does Mia have now?",
|
||
"gold": "13",
|
||
"gen": " 8 + 5 = 13.\n\nLet's double-check: We can verify by counting up from 8: 9, 10, 11, 12, 13. That is 5 counts, so 8 + 5 = 13.\n\nSo",
|
||
"first_line": "8 + 5 = 13.",
|
||
"first_numeric_match": false,
|
||
"final_numeric_match": true,
|
||
"answer_anywhere": true,
|
||
"first_line_answer_anywhere": true
|
||
},
|
||
{
|
||
"problem": "Draw a circle. Then draw a line to cut the circle into two equal shares. How many halves do you have?",
|
||
"gold": "2 halves",
|
||
"gen": " You have one half.\n\nExample 2:\n\n1. Draw a circle.\n2. Draw a line through the center of the circle.\n3. Label the center as A and the line as B.\n\nAnswer",
|
||
"first_line": "You have one half.",
|
||
"first_numeric_match": true,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": true,
|
||
"first_line_answer_anywhere": false
|
||
},
|
||
{
|
||
"problem": "Mia has 8 stickers and gets 6 more stickers. How many stickers does Mia have now?",
|
||
"gold": "14",
|
||
"gen": " 8 + 6 = 14. We can count on from 8: 9, 10, 11, 12, 13, 14. That is 14 counts, so 8 + 6 = 14. We can also use a number line: start at 8",
|
||
"first_line": "8 + 6 = 14. We can count on from 8: 9, 10, 11, 12, 13, 14. That is 14 counts, so 8 + 6 = 14. We can also use a number line: start at 8",
|
||
"first_numeric_match": false,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": true,
|
||
"first_line_answer_anywhere": true
|
||
},
|
||
{
|
||
"problem": "Sarah has 37 stickers. She gets 8 more stickers from her friend. How many stickers does Sarah have now?",
|
||
"gold": "45 stickers",
|
||
"gen": " 37 + 8 = 45.\n\nLet's double-check: We can verify by counting up from 37: 38, 39, 40, 41, 42, 43, 44, 45. That is 8 counts, so we land on",
|
||
"first_line": "37 + 8 = 45.",
|
||
"first_numeric_match": false,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": true,
|
||
"first_line_answer_anywhere": true
|
||
},
|
||
{
|
||
"problem": "Draw a triangle that is red and a triangle that is blue. How are they the same? How are they different?",
|
||
"gold": "They are the same because they both have three sides. They are different because one is red and one is blue.",
|
||
"gen": " They are the same because they both have three sides and three angles. They are different because one is red and one is blue.\n\nExample 2:\n\n* A circle is a shape with all points the same distance from the center",
|
||
"first_line": "They are the same because they both have three sides and three angles. They are different because one is red and one is blue.",
|
||
"first_numeric_match": false,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": false,
|
||
"first_line_answer_anywhere": false
|
||
}
|
||
]
|
||
},
|
||
"corpus_v3/L2/math-7-12-v2/canonical.jsonl": {
|
||
"N": 64,
|
||
"first_numeric_accuracy": 0.015625,
|
||
"final_numeric_accuracy": 0.0,
|
||
"answer_anywhere_rate": 0.078125,
|
||
"first_line_answer_anywhere_rate": 0.0625,
|
||
"ci95_answer_anywhere": [
|
||
0.03383067809543547,
|
||
0.1701975287864777
|
||
],
|
||
"examples": [
|
||
{
|
||
"problem": "A rectangular prism has a length of 8 cm, a width of 5 cm, and a height of 3 cm. What is its surface area?",
|
||
"gold": "158",
|
||
"gen": " Surface area of a rectangular prism is the sum of the areas of all six faces. The formula is 2lw + 2lh + 2wh, where l = length, w = width, h = height. Substituting the given dimensions:",
|
||
"first_line": "Surface area of a rectangular prism is the sum of the areas of all six faces. The formula is 2lw + 2lh + 2wh, where l = length, w = width, h = height. Substitut",
|
||
"first_numeric_match": false,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": false,
|
||
"first_line_answer_anywhere": false
|
||
},
|
||
{
|
||
"problem": "A recipe for homemade lemonade calls for 3/4 cup of lemon juice for every 2 1/2 cups of water. What is the unit rate of cups of lemon juice per cup of water?",
|
||
"gold": "3/10 cup of lemon juice per cup of water",
|
||
"gen": " The unit rate is 3/4 cup of lemon juice per 1 cup of water. We can verify this by checking that the ratio of lemon juice to water is consistent: for 3/4 cup of lemon juice, for 2 1/2",
|
||
"first_line": "The unit rate is 3/4 cup of lemon juice per 1 cup of water. We can verify this by checking that the ratio of lemon juice to water is consistent: for 3/4 cup of ",
|
||
"first_numeric_match": false,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": false,
|
||
"first_line_answer_anywhere": false
|
||
},
|
||
{
|
||
"problem": "A school wants to estimate the average number of hours per week that its 800 students spend on homework. They randomly select 40 students and ask them. The sample average is 6.2 hours with a standard deviation of 1.5 hou",
|
||
"gold": "More reliable, because a larger sample size reduces variability and gives a more precise estimate of the population average.",
|
||
"gen": " The estimate of the population average is more reliable because a larger sample size reduces the chance that the sample is biased or unrepresentative of the population. A larger sample size would allow for greater precision in estimating the population averag",
|
||
"first_line": "The estimate of the population average is more reliable because a larger sample size reduces the chance that the sample is biased or unrepresentative of the pop",
|
||
"first_numeric_match": false,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": false,
|
||
"first_line_answer_anywhere": false
|
||
},
|
||
{
|
||
"problem": "The table shows the number of cups of flour needed to make a certain number of batches of cookies. Determine the constant of proportionality (cups per batch) and write an equation relating cups (c) to batches (b).\n\n| Bat",
|
||
"gold": "Constant of proportionality = 2.5 cups per batch; equation: c = 2.5b",
|
||
"gen": " The constant of proportionality is 5 cups per batch. The equation is c = 5b.\nTo make sure, check it another way: Check by using the equation c = 5b. For b = 4, c = 5 * 4",
|
||
"first_line": "The constant of proportionality is 5 cups per batch. The equation is c = 5b.",
|
||
"first_numeric_match": false,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": false,
|
||
"first_line_answer_anywhere": false
|
||
},
|
||
{
|
||
"problem": "Simplify the expression: 3(2x + 4) - 2(3x - 1).",
|
||
"gold": "14",
|
||
"gen": " 6x + 12 - 6x + 2\n\nLet's double-check: To verify, we can simplify the expression by combining like terms. We can rewrite the second expression as 2x - 1. Now, we can subtract 2",
|
||
"first_line": "6x + 12 - 6x + 2",
|
||
"first_numeric_match": false,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": false,
|
||
"first_line_answer_anywhere": false
|
||
},
|
||
{
|
||
"problem": "A graph shows a line passing through the origin and the point (4, 10). What is the constant of proportionality (unit rate) represented by this graph?",
|
||
"gold": "2.5",
|
||
"gen": " The constant of proportionality is the slope of the line, which is 10. The answer is 10.\nI'm confident in this answer.",
|
||
"first_line": "The constant of proportionality is the slope of the line, which is 10. The answer is 10.",
|
||
"first_numeric_match": false,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": false,
|
||
"first_line_answer_anywhere": false
|
||
}
|
||
]
|
||
},
|
||
"corpus_v3/L2/math_lite_naturalistic_v1/canonical.jsonl": {
|
||
"N": 0,
|
||
"first_numeric_accuracy": 0.0,
|
||
"final_numeric_accuracy": 0.0,
|
||
"answer_anywhere_rate": 0.0,
|
||
"first_line_answer_anywhere_rate": 0.0,
|
||
"ci95_answer_anywhere": [
|
||
0.0,
|
||
0.0
|
||
],
|
||
"examples": []
|
||
},
|
||
"corpus_v3/L2/gold_k6_algebraic_thinking_v1/canonical.jsonl": {
|
||
"N": 64,
|
||
"first_numeric_accuracy": 0.046875,
|
||
"final_numeric_accuracy": 0.0625,
|
||
"answer_anywhere_rate": 0.078125,
|
||
"first_line_answer_anywhere_rate": 0.078125,
|
||
"ci95_answer_anywhere": [
|
||
0.03383067809543547,
|
||
0.1701975287864777
|
||
],
|
||
"examples": [
|
||
{
|
||
"problem": "Use the table or number-line information. A growing pattern begins with term table [(1, 12), (2, 18), (3, 24)]. Find term 4. Check by substituting the answer back.",
|
||
"gold": "30.",
|
||
"gen": " 4\nConfidence: high\nConfidence rationale: The problem uses the same primitive (finding the term in a geometric sequence) with different numbers and a different starting term, but the underlying procedure and reasoning are identical. The verification provides",
|
||
"first_line": "4",
|
||
"first_numeric_match": false,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": false,
|
||
"first_line_answer_anywhere": false
|
||
},
|
||
{
|
||
"problem": "Use the table or number-line information. A pattern has rule value = 6n + 5. Which term number has value 47? Explain the inverse operation you use.",
|
||
"gold": "7.",
|
||
"gen": " The value of n is 47. Use the inverse operation: n=47 gives 6*47+5=223, matching the table.\n\nSo the answer is 47..\nI'm confident in this answer.",
|
||
"first_line": "The value of n is 47. Use the inverse operation: n=47 gives 6*47+5=223, matching the table.",
|
||
"first_numeric_match": false,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": false,
|
||
"first_line_answer_anywhere": false
|
||
},
|
||
{
|
||
"problem": "Use the equation information. A pattern has rule value = 2n + 1. Which term number has value 9? Check by substituting the answer back.",
|
||
"gold": "4.",
|
||
"gen": " 9 has value 9.\nI'm confident in this answer.",
|
||
"first_line": "9 has value 9.",
|
||
"first_numeric_match": false,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": false,
|
||
"first_line_answer_anywhere": false
|
||
},
|
||
{
|
||
"problem": "Use the equation information. Find y so the equality is true: 10 * 4 = 10 * y. Check the answer in the original equation.",
|
||
"gold": "4.",
|
||
"gen": " 10y = 10x + 60. The answer is 10y.\nI'm confident in this answer.",
|
||
"first_line": "10y = 10x + 60. The answer is 10y.",
|
||
"first_numeric_match": false,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": false,
|
||
"first_line_answer_anywhere": false
|
||
},
|
||
{
|
||
"problem": "Use the model information. Use a simple model for this situation: A balance model has left pan 2x + 7 and right pan 43 using beads. Find the missing number. Explain the inverse operation you use. Use x for the unknown, b",
|
||
"gold": "x = 18.",
|
||
"gen": " The missing number is 3. Use x for the unknown, but keep the explanation informal. Check that the answer still matches the model.\n\nUnless required by applicable, use data table or number line approach.\n\n3 × 11",
|
||
"first_line": "The missing number is 3. Use x for the unknown, but keep the explanation informal. Check that the answer still matches the model.",
|
||
"first_numeric_match": false,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": false,
|
||
"first_line_answer_anywhere": false
|
||
},
|
||
{
|
||
"problem": "Use the story information. Rosa starts with 6 books. After giving away an unknown number, 3 are left. Write an equation and find the missing number. Check by substituting the answer back. Use x for the unknown, but keep ",
|
||
"gold": "x = 3.",
|
||
"gen": " The number of books given away is 3.\nConfidence: high\nConfidence rationale: The problem uses the same subtraction-by-a-number procedure with a clear real-world context, and the verification by substitution confirms the answer",
|
||
"first_line": "The number of books given away is 3.",
|
||
"first_numeric_match": true,
|
||
"final_numeric_match": true,
|
||
"answer_anywhere": true,
|
||
"first_line_answer_anywhere": true
|
||
}
|
||
]
|
||
}
|
||
},
|
||
"overall": {
|
||
"first_numeric_accuracy": 0.1015625,
|
||
"final_numeric_accuracy": 0.12890625,
|
||
"answer_anywhere_rate": 0.25390625,
|
||
"first_line_answer_anywhere_rate": 0.23828125,
|
||
"ci95_answer_anywhere": [
|
||
0.20449781341092277,
|
||
0.31059137938961423
|
||
]
|
||
},
|
||
"examples": [
|
||
{
|
||
"source": "corpus_v3/L2/math-drill-v1/canonical.jsonl",
|
||
"problem": "Add 7 and 2.",
|
||
"gold": "9",
|
||
"gen": " The sum is 14. The failure mode to avoid is assuming the answer must be positive without verifying the equation's structure.\nConfidence: 0.85",
|
||
"first_line": "The sum is 14. The failure mode to avoid is assuming the answer must be positive without verifying the equation's structure.",
|
||
"first_numeric_match": false,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": false,
|
||
"first_line_answer_anywhere": false
|
||
},
|
||
{
|
||
"source": "corpus_v3/L2/math-drill-v1/canonical.jsonl",
|
||
"problem": "83 + 49 =",
|
||
"gold": "132",
|
||
"gen": " 50\n\nSo, the sum of the first and last numbers in the sequence is 50.\n\nExample 2: Find the sum of the first five terms of the sequence:\n\n1, 3, 5, 7, 9\n",
|
||
"first_line": "50",
|
||
"first_numeric_match": false,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": false,
|
||
"first_line_answer_anywhere": false
|
||
},
|
||
{
|
||
"source": "corpus_v3/L2/math-drill-v1/canonical.jsonl",
|
||
"problem": "What's 23 plus 91?",
|
||
"gold": "114",
|
||
"gen": " 23 + 91 = 124\n\nSo, the sum of the first and last numbers in our sequence is 124.\n\nNow, let's try another example. Suppose we have the sequence: 1, 2, 3, 4, 5",
|
||
"first_line": "23 + 91 = 124",
|
||
"first_numeric_match": false,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": false,
|
||
"first_line_answer_anywhere": false
|
||
},
|
||
{
|
||
"source": "corpus_v3/L2/math-drill-v1/canonical.jsonl",
|
||
"problem": "8 minus 1 equals what?",
|
||
"gold": "7",
|
||
"gen": " 5\n\nSo, the answer is 5.\nI'm confident in this answer.",
|
||
"first_line": "5",
|
||
"first_numeric_match": false,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": false,
|
||
"first_line_answer_anywhere": false
|
||
},
|
||
{
|
||
"source": "corpus_v3/L2/math-drill-v1/canonical.jsonl",
|
||
"problem": "Divide: 65 ÷ 11",
|
||
"gold": "5 remainder 10",
|
||
"gen": " 65 × 11 = 545\n\nLet's double-check: Verification by reverse division: if 65 × 11 = 544, then 544 ÷ 11 should equal 65. Indeed 544 ÷ 11 = 65. This confirms 65 × 11 = 544.",
|
||
"first_line": "65 × 11 = 545",
|
||
"first_numeric_match": false,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": false,
|
||
"first_line_answer_anywhere": false
|
||
},
|
||
{
|
||
"source": "corpus_v3/L2/math-drill-v1/canonical.jsonl",
|
||
"problem": "? × 3 = 18",
|
||
"gold": "6",
|
||
"gen": " 18\n\nSo, the product of the first and second numbers is 18.\n\nLet's double-check: We can verify this by using the distributive property: The product of the first and third numbers is 12 × 3 = 36",
|
||
"first_line": "18",
|
||
"first_numeric_match": false,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": false,
|
||
"first_line_answer_anywhere": false
|
||
},
|
||
{
|
||
"source": "corpus_v3/L2/math-drill-v1/canonical.jsonl",
|
||
"problem": "Sum 780 and 389.",
|
||
"gold": "1169",
|
||
"gen": " The sum of the digits of the number is 7 + 8 = 15.\n\nExample 2: Find the sum of the digits of the number 12,343.\nAnswer: The sum of the digits of the number is 4 +",
|
||
"first_line": "The sum of the digits of the number is 7 + 8 = 15.",
|
||
"first_numeric_match": false,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": false,
|
||
"first_line_answer_anywhere": false
|
||
},
|
||
{
|
||
"source": "corpus_v3/L2/math-drill-v1/canonical.jsonl",
|
||
"problem": "Compute the sum: 236 + 328.",
|
||
"gold": "564",
|
||
"gen": " 238\n\nLet's double-check: Verification by reverse subtraction: if 238 - 512 = 32, then 32 + 512 should equal 238. Indeed 32 + 512 = 238. This confirms 238 - 512 = 32.\n\nSo the",
|
||
"first_line": "238",
|
||
"first_numeric_match": false,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": false,
|
||
"first_line_answer_anywhere": false
|
||
},
|
||
{
|
||
"source": "corpus_v3/L2/math-drill-v1/canonical.jsonl",
|
||
"problem": "What's 32 plus 24?",
|
||
"gold": "56",
|
||
"gen": " 64 + 24 = 100\n\nSo, the sum of the first and last numbers is 100.\n\nLet's double-check: We can verify by grouping the numbers differently. If we take the middle number (4) and add",
|
||
"first_line": "64 + 24 = 100",
|
||
"first_numeric_match": false,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": false,
|
||
"first_line_answer_anywhere": false
|
||
},
|
||
{
|
||
"source": "corpus_v3/L2/math-drill-v1/canonical.jsonl",
|
||
"problem": "Evaluate 939 + 350.",
|
||
"gold": "1289",
|
||
"gen": " 5,450\nConfidence: high\nConfidence rationale: The problem uses the same primitive operation (addition of decimals with like denominators) as the original, only changing the numbers and the operation. The verification via decimal conversion confirms",
|
||
"first_line": "5,450",
|
||
"first_numeric_match": false,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": false,
|
||
"first_line_answer_anywhere": false
|
||
},
|
||
{
|
||
"source": "corpus_v3/L2/math-drill-v1/canonical.jsonl",
|
||
"problem": "7 + 7 = ?",
|
||
"gold": "14",
|
||
"gen": " 14\n\nSo, the sum of the first and last numbers in the sequence is 14.\n\nLet's double-check: We can verify this by grouping the numbers differently. Notice that the middle number is 7. The first",
|
||
"first_line": "14",
|
||
"first_numeric_match": true,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": true,
|
||
"first_line_answer_anywhere": true
|
||
},
|
||
{
|
||
"source": "corpus_v3/L2/math-drill-v1/canonical.jsonl",
|
||
"problem": "Divide 16 by 9.",
|
||
"gold": "1 remainder 7",
|
||
"gen": " 16 divided by 9 is 1 remainder 4. So each bus carries 4 students. The answer is 4.\nI'm confident in this answer.",
|
||
"first_line": "16 divided by 9 is 1 remainder 4. So each bus carries 4 students. The answer is 4.",
|
||
"first_numeric_match": false,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": false,
|
||
"first_line_answer_anywhere": false
|
||
},
|
||
{
|
||
"source": "corpus_v3/L2/math-drill-v1/canonical.jsonl",
|
||
"problem": "What is 4 × 7?",
|
||
"gold": "28",
|
||
"gen": " 28\n\nSo the product is 28.\nTo make sure, check it another way: Verification by reverse division: if 4 × 7 = 28, then 28 ÷ 7 should equal 4. Indeed 28 ÷ 7 = 4. This confirms",
|
||
"first_line": "28",
|
||
"first_numeric_match": true,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": true,
|
||
"first_line_answer_anywhere": true
|
||
},
|
||
{
|
||
"source": "corpus_v3/L2/math-drill-v1/canonical.jsonl",
|
||
"problem": "11 × 9 = ?",
|
||
"gold": "99",
|
||
"gen": " 81\n\nSo, the product of the first and second numbers is 81.\n\nLet's double-check: We can verify this by using the distributive property: The product of the first and third numbers is 3 × 9 = 27",
|
||
"first_line": "81",
|
||
"first_numeric_match": false,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": false,
|
||
"first_line_answer_anywhere": false
|
||
},
|
||
{
|
||
"source": "corpus_v3/L2/math-drill-v1/canonical.jsonl",
|
||
"problem": "Compute 7 × 6.",
|
||
"gold": "42",
|
||
"gen": " 42\nConfidence: high\nConfidence rationale: The problem uses the same primitive operation (multiplication of two-digit numbers) with a different number of digits and a different operation (division instead of addition). The reasoning and verification follow",
|
||
"first_line": "42",
|
||
"first_numeric_match": true,
|
||
"final_numeric_match": true,
|
||
"answer_anywhere": true,
|
||
"first_line_answer_anywhere": true
|
||
},
|
||
{
|
||
"source": "corpus_v3/L2/math-drill-v1/canonical.jsonl",
|
||
"problem": "Compute 68 + 92.",
|
||
"gold": "160",
|
||
"gen": " 610\nConfidence: high\nConfidence rationale: The problem uses the same primitive operation (addition of two-digit numbers) with a different base and operation, and the verification via estimation confirms the result.",
|
||
"first_line": "610",
|
||
"first_numeric_match": false,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": false,
|
||
"first_line_answer_anywhere": false
|
||
},
|
||
{
|
||
"source": "corpus_v3/L2/math-drill-v1/canonical.jsonl",
|
||
"problem": "Solve: 198 + 903",
|
||
"gold": "1101",
|
||
"gen": " 200\n\nLet's double-check: Verification by reverse subtraction: if 198 + 702 = 200, then 200 - 702 should equal 198. Indeed 200 - 702 = 198. This confirms 198 + 702 = 200.\n\nSo the",
|
||
"first_line": "200",
|
||
"first_numeric_match": false,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": false,
|
||
"first_line_answer_anywhere": false
|
||
},
|
||
{
|
||
"source": "corpus_v3/L2/math-drill-v1/canonical.jsonl",
|
||
"problem": "Add: 4244 + 3127",
|
||
"gold": "7371",
|
||
"gen": " 68\n\nSo, the sum of the first and last numbers in the sequence is 68.\n\nLet's double-check: We can verify this by grouping the numbers differently. Notice that the middle number is the average of the",
|
||
"first_line": "68",
|
||
"first_numeric_match": false,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": false,
|
||
"first_line_answer_anywhere": false
|
||
},
|
||
{
|
||
"source": "corpus_v3/L2/math-drill-v1/canonical.jsonl",
|
||
"problem": "What is 9 × 12?",
|
||
"gold": "108",
|
||
"gen": " 108\n\nLet's double-check: Verification by reverse division: if 9 × 12 = 108, then 108 ÷ 12 should equal 9. Indeed 108 ÷ 12 = 9. This confirms 9 × 12 = 108.\n\nSo the",
|
||
"first_line": "108",
|
||
"first_numeric_match": true,
|
||
"final_numeric_match": true,
|
||
"answer_anywhere": true,
|
||
"first_line_answer_anywhere": true
|
||
},
|
||
{
|
||
"source": "corpus_v3/L2/math-drill-v1/canonical.jsonl",
|
||
"problem": "Divide: 18 ÷ 8",
|
||
"gold": "2 remainder 2",
|
||
"gen": " 18 ÷ 8 = 2 remainder 2, so 18 ÷ 8 = 2 2/8. The answer is 2 2/8.\nI'm confident in this answer.",
|
||
"first_line": "18 ÷ 8 = 2 remainder 2, so 18 ÷ 8 = 2 2/8. The answer is 2 2/8.",
|
||
"first_numeric_match": false,
|
||
"final_numeric_match": false,
|
||
"answer_anywhere": true,
|
||
"first_line_answer_anywhere": true
|
||
}
|
||
]
|
||
},
|
||
"code_first_function": {
|
||
"N": 6,
|
||
"ast_rate": 1.0,
|
||
"non_docstring_body_rate": 0.0,
|
||
"return_rate": 0.0,
|
||
"exec_pass_rate": 0.0,
|
||
"ci95_exec": [
|
||
0.0,
|
||
0.3903430336530645
|
||
],
|
||
"examples": [
|
||
{
|
||
"name": "add_one",
|
||
"prompt": "def add_one(n):",
|
||
"generated": "\ndef test_add_one():\n \"\"\"Test add_one.\"\"\"\n assert add_one(1) == 2\n\ndef test_add_two():\n \"\"\"Test add_two.\"\"\"\n assert add_two(1) == 2\n\ndef test_add_three():\n \"\"\"Test add_three.\"\"\"\n assert add_three(1) == 3\n\ndef test_add_four():\n \"\"\"Test add_four.\"\"\"\n assert add",
|
||
"first_function": "def add_one(n):\n \"\"\"Return n plus one.\"\"\"\n",
|
||
"ast_ok": true,
|
||
"has_non_docstring_body": false,
|
||
"has_return": false,
|
||
"undefined_names": [],
|
||
"exec_ok": false
|
||
},
|
||
{
|
||
"name": "reverse_string",
|
||
"prompt": "def reverse_string(s):",
|
||
"generated": "",
|
||
"first_function": "def reverse_string(s):\n \"\"\"Return s reversed.\"\"\"\n",
|
||
"ast_ok": true,
|
||
"has_non_docstring_body": false,
|
||
"has_return": false,
|
||
"undefined_names": [],
|
||
"exec_ok": false
|
||
},
|
||
{
|
||
"name": "count_vowels",
|
||
"prompt": "def count_vowels(s):",
|
||
"generated": "\ndef count_digits(s):\n \"\"\"Return the number of digits in s.\"\"\"\n\ndef count_all_digits(s):\n \"\"\"Return the number of digits in s.\"\"\"\n\ndef count_all_letters(s):\n \"\"\"Return the number of letters in s.\"\"\"\n\ndef count_all_digits(s):\n \"\"\"Return the number of digits in s.\"\"\"\n\ndef count_all_letters(s):\n ",
|
||
"first_function": "def count_vowels(s):\n \"\"\"Return the number of vowels in s.\"\"\"\n",
|
||
"ast_ok": true,
|
||
"has_non_docstring_body": false,
|
||
"has_return": false,
|
||
"undefined_names": [],
|
||
"exec_ok": false
|
||
},
|
||
{
|
||
"name": "is_even",
|
||
"prompt": "def is_even(n):",
|
||
"generated": "else:\n is_even = lambda n: n % 2 == 0\n\ndef is_odd(n):\n \"\"\"Return True if n is odd.\"\"\"\nelse:\n is_odd = lambda n: n % 2 == 0\n\ndef is_zero(n):\n \"\"\"Return True if n is zero.\"\"\"\nelse:\n is_zero = lambda n: n == 0\n\ndef is_even(n):\n \"\"\"Return True if n",
|
||
"first_function": "def is_even(n):\n \"\"\"Return True if n is even.\"\"\"\n",
|
||
"ast_ok": true,
|
||
"has_non_docstring_body": false,
|
||
"has_return": false,
|
||
"undefined_names": [],
|
||
"exec_ok": false
|
||
},
|
||
{
|
||
"name": "sum_list",
|
||
"prompt": "def sum_list(items):",
|
||
"generated": "\ndef get_sum_list(items):\n \"\"\"Return the sum of the numbers in items.\"\"\"\n\ndef get_sum_list_from_list(items):\n \"\"\"Return the sum of the numbers in items from the list.\"\"\"\n\ndef get_sum_list_from_list_with_prefix(items):\n \"\"\"Return the sum of the numbers in items from the list with the prefix.\"\"\"\n\ndef get_sum_list_from",
|
||
"first_function": "def sum_list(items):\n \"\"\"Return the sum of the numbers in items.\"\"\"\n",
|
||
"ast_ok": true,
|
||
"has_non_docstring_body": false,
|
||
"has_return": false,
|
||
"undefined_names": [],
|
||
"exec_ok": false
|
||
},
|
||
{
|
||
"name": "first_item",
|
||
"prompt": "def first_item(items):",
|
||
"generated": "\ndef get_first_item(items):\n \"\"\"Return the first item in a non-empty list.\"\"\"\n\ndef get_first_item_from_list(items):\n \"\"\"Return the first item from a non-empty list.\"\"\"\n\ndef get_first_item_from_list_of_strings(items):\n \"\"\"Return the first item from a list of strings.\"\"\"\n\ndef get_first_item_from_list_of",
|
||
"first_function": "def first_item(items):\n \"\"\"Return the first item in a non-empty list.\"\"\"\n",
|
||
"ast_ok": true,
|
||
"has_non_docstring_body": false,
|
||
"has_return": false,
|
||
"undefined_names": [],
|
||
"exec_ok": false
|
||
}
|
||
]
|
||
},
|
||
"corpus_direct_answer_audit": {
|
||
"manifest": "runs/v10_348m_cone/manifest_live.yaml",
|
||
"by_component": [
|
||
{
|
||
"id": "L0_tom_semantic_topup_v1",
|
||
"layer": "L0",
|
||
"N": 5000,
|
||
"output_answer_rate": 0.0,
|
||
"answer_key_rate": 0.0,
|
||
"answer_marker_rate": 0.0
|
||
},
|
||
{
|
||
"id": "L0_tom_semantic_equalizer_v1",
|
||
"layer": "L0",
|
||
"N": 4336,
|
||
"output_answer_rate": 0.0,
|
||
"answer_key_rate": 0.0,
|
||
"answer_marker_rate": 0.0
|
||
},
|
||
{
|
||
"id": "L0_5_cosmopedia_l1_reroute_v1",
|
||
"layer": "L0_5",
|
||
"N": 5000,
|
||
"output_answer_rate": 0.0,
|
||
"answer_key_rate": 0.0,
|
||
"answer_marker_rate": 0.0012
|
||
},
|
||
{
|
||
"id": "L1_naturalistic_reasoning_v2",
|
||
"layer": "L1",
|
||
"N": 5000,
|
||
"output_answer_rate": 0.0,
|
||
"answer_key_rate": 0.0,
|
||
"answer_marker_rate": 0.0016
|
||
},
|
||
{
|
||
"id": "L1_naturalistic_reasoning_v1",
|
||
"layer": "L1",
|
||
"N": 5000,
|
||
"output_answer_rate": 0.0,
|
||
"answer_key_rate": 0.0,
|
||
"answer_marker_rate": 0.0022
|
||
},
|
||
{
|
||
"id": "L1_logic-primitives-v2",
|
||
"layer": "L1",
|
||
"N": 5000,
|
||
"output_answer_rate": 1.0,
|
||
"answer_key_rate": 0.0,
|
||
"answer_marker_rate": 0.0062
|
||
},
|
||
{
|
||
"id": "L1_logic-primitives-v1",
|
||
"layer": "L1",
|
||
"N": 5000,
|
||
"output_answer_rate": 1.0,
|
||
"answer_key_rate": 0.0,
|
||
"answer_marker_rate": 0.002
|
||
},
|
||
{
|
||
"id": "L1_logic-books-bulk-v1",
|
||
"layer": "L1",
|
||
"N": 918,
|
||
"output_answer_rate": 0.0,
|
||
"answer_key_rate": 0.0,
|
||
"answer_marker_rate": 0.007625272331154684
|
||
},
|
||
{
|
||
"id": "L1_cac-compact-v1",
|
||
"layer": "L1",
|
||
"N": 500,
|
||
"output_answer_rate": 1.0,
|
||
"answer_key_rate": 0.0,
|
||
"answer_marker_rate": 0.0
|
||
},
|
||
{
|
||
"id": "L2_math-7-12-v2",
|
||
"layer": "L2",
|
||
"N": 5000,
|
||
"output_answer_rate": 1.0,
|
||
"answer_key_rate": 0.0,
|
||
"answer_marker_rate": 0.0002
|
||
},
|
||
{
|
||
"id": "L2_math-k6-v2",
|
||
"layer": "L2",
|
||
"N": 5000,
|
||
"output_answer_rate": 1.0,
|
||
"answer_key_rate": 0.0,
|
||
"answer_marker_rate": 0.0028
|
||
},
|
||
{
|
||
"id": "L2_math_lite_naturalistic_v1",
|
||
"layer": "L2",
|
||
"N": 5000,
|
||
"output_answer_rate": 0.0,
|
||
"answer_key_rate": 0.0,
|
||
"answer_marker_rate": 0.0
|
||
},
|
||
{
|
||
"id": "L2_math-drill-v1",
|
||
"layer": "L2",
|
||
"N": 5000,
|
||
"output_answer_rate": 1.0,
|
||
"answer_key_rate": 0.0,
|
||
"answer_marker_rate": 0.0
|
||
},
|
||
{
|
||
"id": "L2_gold_k6_algebraic_thinking_v1",
|
||
"layer": "L2",
|
||
"N": 4800,
|
||
"output_answer_rate": 1.0,
|
||
"answer_key_rate": 1.0,
|
||
"answer_marker_rate": 0.0
|
||
},
|
||
{
|
||
"id": "L2_entry_subjects_gold_v1",
|
||
"layer": "L2",
|
||
"N": 5000,
|
||
"output_answer_rate": 0.0,
|
||
"answer_key_rate": 0.0,
|
||
"answer_marker_rate": 0.0004
|
||
},
|
||
{
|
||
"id": "L2_intro_books_gold_clean_v1",
|
||
"layer": "L2",
|
||
"N": 2634,
|
||
"output_answer_rate": 0.0,
|
||
"answer_key_rate": 0.0,
|
||
"answer_marker_rate": 0.013287775246772968
|
||
},
|
||
{
|
||
"id": "L2_math-bulk-v1",
|
||
"layer": "L2",
|
||
"N": 196,
|
||
"output_answer_rate": 0.0,
|
||
"answer_key_rate": 0.0,
|
||
"answer_marker_rate": 0.01020408163265306
|
||
},
|
||
{
|
||
"id": "L2_algebra-bulk-v1",
|
||
"layer": "L2",
|
||
"N": 369,
|
||
"output_answer_rate": 0.0,
|
||
"answer_key_rate": 0.0,
|
||
"answer_marker_rate": 0.02168021680216802
|
||
},
|
||
{
|
||
"id": "L2_geometry-bulk-v1",
|
||
"layer": "L2",
|
||
"N": 172,
|
||
"output_answer_rate": 0.0,
|
||
"answer_key_rate": 0.0,
|
||
"answer_marker_rate": 0.040697674418604654
|
||
},
|
||
{
|
||
"id": "L2_trig-bulk-v1",
|
||
"layer": "L2",
|
||
"N": 111,
|
||
"output_answer_rate": 0.0,
|
||
"answer_key_rate": 0.0,
|
||
"answer_marker_rate": 0.0
|
||
},
|
||
{
|
||
"id": "L2_precalculus-source-bulk-v1",
|
||
"layer": "L2",
|
||
"N": 124,
|
||
"output_answer_rate": 0.0,
|
||
"answer_key_rate": 0.0,
|
||
"answer_marker_rate": 0.016129032258064516
|
||
},
|
||
{
|
||
"id": "primer_math",
|
||
"layer": "L2",
|
||
"N": 170,
|
||
"output_answer_rate": 1.0,
|
||
"answer_key_rate": 0.0,
|
||
"answer_marker_rate": 0.0
|
||
},
|
||
{
|
||
"id": "L3_python_no_eval_overlap_clean_v1",
|
||
"layer": "L3",
|
||
"N": 5000,
|
||
"output_answer_rate": 0.0,
|
||
"answer_key_rate": 0.0,
|
||
"answer_marker_rate": 0.005
|
||
},
|
||
{
|
||
"id": "L3_code_lane_clean_v3",
|
||
"layer": "L3",
|
||
"N": 5000,
|
||
"output_answer_rate": 0.0,
|
||
"answer_key_rate": 0.0,
|
||
"answer_marker_rate": 0.0044
|
||
},
|
||
{
|
||
"id": "L3_process_reasoning_books_v1",
|
||
"layer": "L3",
|
||
"N": 797,
|
||
"output_answer_rate": 0.0,
|
||
"answer_key_rate": 0.0,
|
||
"answer_marker_rate": 0.002509410288582183
|
||
},
|
||
{
|
||
"id": "L3_algorithms-bulk-v1",
|
||
"layer": "L3",
|
||
"N": 298,
|
||
"output_answer_rate": 0.0,
|
||
"answer_key_rate": 0.0,
|
||
"answer_marker_rate": 0.006711409395973154
|
||
},
|
||
{
|
||
"id": "L3_data-science-bulk-clean-v1",
|
||
"layer": "L3",
|
||
"N": 294,
|
||
"output_answer_rate": 0.0,
|
||
"answer_key_rate": 0.0,
|
||
"answer_marker_rate": 0.0
|
||
},
|
||
{
|
||
"id": "L3_algorithms-books-new-clean-v1",
|
||
"layer": "L3",
|
||
"N": 35,
|
||
"output_answer_rate": 0.0,
|
||
"answer_key_rate": 0.0,
|
||
"answer_marker_rate": 0.0
|
||
},
|
||
{
|
||
"id": "L3_machine-learning-books-bulk-v1",
|
||
"layer": "L3",
|
||
"N": 6,
|
||
"output_answer_rate": 0.0,
|
||
"answer_key_rate": 0.0,
|
||
"answer_marker_rate": 0.0
|
||
},
|
||
{
|
||
"id": "primer_code_disciplines",
|
||
"layer": "L3",
|
||
"N": 620,
|
||
"output_answer_rate": 1.0,
|
||
"answer_key_rate": 0.0,
|
||
"answer_marker_rate": 0.0
|
||
},
|
||
{
|
||
"id": "agentic_agent-flan-v1",
|
||
"layer": "L4",
|
||
"N": 5000,
|
||
"output_answer_rate": 0.0,
|
||
"answer_key_rate": 0.0,
|
||
"answer_marker_rate": 0.0816
|
||
},
|
||
{
|
||
"id": "primer_agentic",
|
||
"layer": "L4",
|
||
"N": 125,
|
||
"output_answer_rate": 1.0,
|
||
"answer_key_rate": 0.0,
|
||
"answer_marker_rate": 0.016
|
||
}
|
||
],
|
||
"by_layer": {
|
||
"L0": {
|
||
"N": 9336,
|
||
"output_answer_rate": 0.0,
|
||
"answer_key_rate": 0.0,
|
||
"answer_marker_rate": 0.0
|
||
},
|
||
"L0_5": {
|
||
"N": 5000,
|
||
"output_answer_rate": 0.0,
|
||
"answer_key_rate": 0.0,
|
||
"answer_marker_rate": 0.0012
|
||
},
|
||
"L1": {
|
||
"N": 21418,
|
||
"output_answer_rate": 0.490241852647306,
|
||
"answer_key_rate": 0.0,
|
||
"answer_marker_rate": 0.0031282099168923336
|
||
},
|
||
"L2": {
|
||
"N": 33576,
|
||
"output_answer_rate": 0.5947700738622825,
|
||
"answer_key_rate": 0.14295925661186562,
|
||
"answer_marker_rate": 0.002114605670717179
|
||
},
|
||
"L3": {
|
||
"N": 12050,
|
||
"output_answer_rate": 0.051452282157676346,
|
||
"answer_key_rate": 0.0,
|
||
"answer_marker_rate": 0.004232365145228216
|
||
},
|
||
"L4": {
|
||
"N": 5125,
|
||
"output_answer_rate": 0.024390243902439025,
|
||
"answer_key_rate": 0.0,
|
||
"answer_marker_rate": 0.08
|
||
}
|
||
}
|
||
}
|
||
},
|
||
"status": "complete"
|
||
}
|