[{"data":1,"prerenderedAt":2425},["ShallowReactive",2],{"page-\u002Fprompt-engineering\u002F05-chain-of-thought-prompting":3},{"id":4,"title":5,"body":6,"description":2418,"extension":2419,"meta":2420,"navigation":128,"path":2421,"seo":2422,"stem":2423,"__hash__":2424},"content\u002Fprompt-engineering\u002F05-chain-of-thought-prompting.md","05 — Chain-of-Thought Prompting",{"type":7,"value":8,"toc":2405},"minimark",[9,13,18,288,292,326,368,372,457,461,816,820,1008,1012,1045,1488,1492,1543,1954,1958,2067,2071,2225,2229,2236,2266,2270,2401],[10,11,5],"h1",{"id":12},"_05-chain-of-thought-prompting",[14,15,17],"h2",{"id":16},"the-mechanism-cot-as-self-generated-context","The Mechanism: CoT as Self-Generated Context",[19,20,23],"code-wrapper",{"filename":21,"language":22},"cot_mechanism.py","python",[24,25,29],"pre",{"className":26,"code":27,"language":22,"meta":28,"style":28},"language-python shiki shiki-themes github-light github-dark","# WHY CoT WORKS — the autoregressive mechanism:\n#\n# WITHOUT CoT: the model must arrive at the correct answer in ONE shot,\n# with no intermediate \"scratch space.\" The final answer token is conditioned\n# only on the prompt + whatever implicit internal computation happens in a\n# single forward pass.\n#\n# WITH CoT: the model writes intermediate steps. Those steps become PART OF\n# the context that later tokens condition on. Each step narrows the space\n# of plausible next steps. The model is using its OWN GENERATED TEXT as\n# working memory.\n#\n# CoT doesn't make the model \"think harder\" in some abstract sense — it\n# gives the model more tokens of relevant, self-generated context to\n# condition the final answer on.\n\n# Pseudocode of the difference:\ndef without_cot(prompt):\n    # Answer token conditioned on prompt only\n    answer = model.generate(prompt + \"Answer:\")\n    return answer  # one-shot — no scratch space\n\ndef with_cot(prompt):\n    # Reasoning tokens become context for the answer\n    reasoning = model.generate(prompt + \"Think step by step:\")\n    answer = model.generate(prompt + reasoning + \"Answer:\")\n    return answer  # reasoning tokens are now load-bearing context for the answer\n\n# Each intermediate line is a CHECKPOINT the model can verify against:\n# if the running total is nonsensical (negative, absurdly large), that's\n# visible in the token stream and can influence correction — a single\n# hidden mental step cannot.\n","",[30,31,32,41,47,53,59,65,71,76,82,88,94,100,105,111,117,123,130,136,151,157,179,191,196,206,212,229,249,259,264,270,276,282],"code",{"__ignoreMap":28},[33,34,37],"span",{"class":35,"line":36},"line",1,[33,38,40],{"class":39},"sdCPZ","# WHY CoT WORKS — the autoregressive mechanism:\n",[33,42,44],{"class":35,"line":43},2,[33,45,46],{"class":39},"#\n",[33,48,50],{"class":35,"line":49},3,[33,51,52],{"class":39},"# WITHOUT CoT: the model must arrive at the correct answer in ONE shot,\n",[33,54,56],{"class":35,"line":55},4,[33,57,58],{"class":39},"# with no intermediate \"scratch space.\" The final answer token is conditioned\n",[33,60,62],{"class":35,"line":61},5,[33,63,64],{"class":39},"# only on the prompt + whatever implicit internal computation happens in a\n",[33,66,68],{"class":35,"line":67},6,[33,69,70],{"class":39},"# single forward pass.\n",[33,72,74],{"class":35,"line":73},7,[33,75,46],{"class":39},[33,77,79],{"class":35,"line":78},8,[33,80,81],{"class":39},"# WITH CoT: the model writes intermediate steps. Those steps become PART OF\n",[33,83,85],{"class":35,"line":84},9,[33,86,87],{"class":39},"# the context that later tokens condition on. Each step narrows the space\n",[33,89,91],{"class":35,"line":90},10,[33,92,93],{"class":39},"# of plausible next steps. The model is using its OWN GENERATED TEXT as\n",[33,95,97],{"class":35,"line":96},11,[33,98,99],{"class":39},"# working memory.\n",[33,101,103],{"class":35,"line":102},12,[33,104,46],{"class":39},[33,106,108],{"class":35,"line":107},13,[33,109,110],{"class":39},"# CoT doesn't make the model \"think harder\" in some abstract sense — it\n",[33,112,114],{"class":35,"line":113},14,[33,115,116],{"class":39},"# gives the model more tokens of relevant, self-generated context to\n",[33,118,120],{"class":35,"line":119},15,[33,121,122],{"class":39},"# condition the final answer on.\n",[33,124,126],{"class":35,"line":125},16,[33,127,129],{"emptyLinePlaceholder":128},true,"\n",[33,131,133],{"class":35,"line":132},17,[33,134,135],{"class":39},"# Pseudocode of the difference:\n",[33,137,139,143,147],{"class":35,"line":138},18,[33,140,142],{"class":141},"svdQ7","def",[33,144,146],{"class":145},"sIsaT"," without_cot",[33,148,150],{"class":149},"ssxIu","(prompt):\n",[33,152,154],{"class":35,"line":153},19,[33,155,156],{"class":39},"    # Answer token conditioned on prompt only\n",[33,158,160,163,166,169,172,176],{"class":35,"line":159},20,[33,161,162],{"class":149},"    answer ",[33,164,165],{"class":141},"=",[33,167,168],{"class":149}," model.generate(prompt ",[33,170,171],{"class":141},"+",[33,173,175],{"class":174},"sJ6F3"," \"Answer:\"",[33,177,178],{"class":149},")\n",[33,180,182,185,188],{"class":35,"line":181},21,[33,183,184],{"class":141},"    return",[33,186,187],{"class":149}," answer  ",[33,189,190],{"class":39},"# one-shot — no scratch space\n",[33,192,194],{"class":35,"line":193},22,[33,195,129],{"emptyLinePlaceholder":128},[33,197,199,201,204],{"class":35,"line":198},23,[33,200,142],{"class":141},[33,202,203],{"class":145}," with_cot",[33,205,150],{"class":149},[33,207,209],{"class":35,"line":208},24,[33,210,211],{"class":39},"    # Reasoning tokens become context for the answer\n",[33,213,215,218,220,222,224,227],{"class":35,"line":214},25,[33,216,217],{"class":149},"    reasoning ",[33,219,165],{"class":141},[33,221,168],{"class":149},[33,223,171],{"class":141},[33,225,226],{"class":174}," \"Think step by step:\"",[33,228,178],{"class":149},[33,230,232,234,236,238,240,243,245,247],{"class":35,"line":231},26,[33,233,162],{"class":149},[33,235,165],{"class":141},[33,237,168],{"class":149},[33,239,171],{"class":141},[33,241,242],{"class":149}," reasoning ",[33,244,171],{"class":141},[33,246,175],{"class":174},[33,248,178],{"class":149},[33,250,252,254,256],{"class":35,"line":251},27,[33,253,184],{"class":141},[33,255,187],{"class":149},[33,257,258],{"class":39},"# reasoning tokens are now load-bearing context for the answer\n",[33,260,262],{"class":35,"line":261},28,[33,263,129],{"emptyLinePlaceholder":128},[33,265,267],{"class":35,"line":266},29,[33,268,269],{"class":39},"# Each intermediate line is a CHECKPOINT the model can verify against:\n",[33,271,273],{"class":35,"line":272},30,[33,274,275],{"class":39},"# if the running total is nonsensical (negative, absurdly large), that's\n",[33,277,279],{"class":35,"line":278},31,[33,280,281],{"class":39},"# visible in the token stream and can influence correction — a single\n",[33,283,285],{"class":35,"line":284},32,[33,286,287],{"class":39},"# hidden mental step cannot.\n",[14,289,291],{"id":290},"zero-shot-cot-the-one-line-trigger","Zero-Shot CoT: The One-Line Trigger",[19,293,296],{"filename":294,"language":295},"zero_shot_cot.md","markdown",[24,297,300],{"className":298,"code":299,"language":295,"meta":28,"style":28},"language-markdown shiki shiki-themes github-light github-dark","A store had 142 units of a product. They sold 37% of their stock on Monday,\nthen received a shipment of 60 more units. On Tuesday they sold 28 units.\nHow many units are left?\n\nThink through this step by step before giving your final answer.\n",[30,301,302,307,312,317,321],{"__ignoreMap":28},[33,303,304],{"class":35,"line":36},[33,305,306],{"class":149},"A store had 142 units of a product. They sold 37% of their stock on Monday,\n",[33,308,309],{"class":35,"line":43},[33,310,311],{"class":149},"then received a shipment of 60 more units. On Tuesday they sold 28 units.\n",[33,313,314],{"class":35,"line":49},[33,315,316],{"class":149},"How many units are left?\n",[33,318,319],{"class":35,"line":55},[33,320,129],{"emptyLinePlaceholder":128},[33,322,323],{"class":35,"line":61},[33,324,325],{"class":149},"Think through this step by step before giving your final answer.\n",[19,327,329],{"filename":328,"language":295},"zero_shot_cot_output.md",[24,330,332],{"className":298,"code":331,"language":295,"meta":28,"style":28},"Starting stock: 142 units\nSold 37% on Monday: 142 × 0.37 = 52.54, round to 53 units sold\nRemaining after Monday: 142 - 53 = 89 units\nReceived shipment: 89 + 60 = 149 units\nSold on Tuesday: 149 - 28 = 121 units\n\nFinal answer: 121 units remain.\n",[30,333,334,339,344,349,354,359,363],{"__ignoreMap":28},[33,335,336],{"class":35,"line":36},[33,337,338],{"class":149},"Starting stock: 142 units\n",[33,340,341],{"class":35,"line":43},[33,342,343],{"class":149},"Sold 37% on Monday: 142 × 0.37 = 52.54, round to 53 units sold\n",[33,345,346],{"class":35,"line":49},[33,347,348],{"class":149},"Remaining after Monday: 142 - 53 = 89 units\n",[33,350,351],{"class":35,"line":55},[33,352,353],{"class":149},"Received shipment: 89 + 60 = 149 units\n",[33,355,356],{"class":35,"line":61},[33,357,358],{"class":149},"Sold on Tuesday: 149 - 28 = 121 units\n",[33,360,361],{"class":35,"line":67},[33,362,129],{"emptyLinePlaceholder":128},[33,364,365],{"class":35,"line":73},[33,366,367],{"class":149},"Final answer: 121 units remain.\n",[14,369,371],{"id":370},"few-shot-cot-demonstrating-the-reasoning-pattern","Few-Shot CoT: Demonstrating the Reasoning Pattern",[19,373,375],{"filename":374,"language":295},"few_shot_cot.md",[24,376,378],{"className":298,"code":377,"language":295,"meta":28,"style":28},"Q: A cafe sells cups of coffee for $4 and pastries for $3. On a day they\nsold 45 coffees and 20 pastries, but 3 pastries were returned for a refund,\nwhat was their net revenue?\n\nA: Coffee revenue: 45 × $4 = $180\nPastry revenue before returns: 20 × $3 = $60\nRefunds for returned pastries: 3 × $3 = $9\nNet pastry revenue: $60 - $9 = $51\nTotal net revenue: $180 + $51 = $231\nThe answer is $231.\n\nQ: A parking garage charges $5 for the first hour and $2 for each\nadditional hour. A customer parked for 6 hours but has a coupon for 25%\noff the total. What did they pay?\n\nA:\n",[30,379,380,385,390,395,399,404,409,414,419,424,429,433,438,443,448,452],{"__ignoreMap":28},[33,381,382],{"class":35,"line":36},[33,383,384],{"class":149},"Q: A cafe sells cups of coffee for $4 and pastries for $3. On a day they\n",[33,386,387],{"class":35,"line":43},[33,388,389],{"class":149},"sold 45 coffees and 20 pastries, but 3 pastries were returned for a refund,\n",[33,391,392],{"class":35,"line":49},[33,393,394],{"class":149},"what was their net revenue?\n",[33,396,397],{"class":35,"line":55},[33,398,129],{"emptyLinePlaceholder":128},[33,400,401],{"class":35,"line":61},[33,402,403],{"class":149},"A: Coffee revenue: 45 × $4 = $180\n",[33,405,406],{"class":35,"line":67},[33,407,408],{"class":149},"Pastry revenue before returns: 20 × $3 = $60\n",[33,410,411],{"class":35,"line":73},[33,412,413],{"class":149},"Refunds for returned pastries: 3 × $3 = $9\n",[33,415,416],{"class":35,"line":78},[33,417,418],{"class":149},"Net pastry revenue: $60 - $9 = $51\n",[33,420,421],{"class":35,"line":84},[33,422,423],{"class":149},"Total net revenue: $180 + $51 = $231\n",[33,425,426],{"class":35,"line":90},[33,427,428],{"class":149},"The answer is $231.\n",[33,430,431],{"class":35,"line":96},[33,432,129],{"emptyLinePlaceholder":128},[33,434,435],{"class":35,"line":102},[33,436,437],{"class":149},"Q: A parking garage charges $5 for the first hour and $2 for each\n",[33,439,440],{"class":35,"line":107},[33,441,442],{"class":149},"additional hour. A customer parked for 6 hours but has a coupon for 25%\n",[33,444,445],{"class":35,"line":113},[33,446,447],{"class":149},"off the total. What did they pay?\n",[33,449,450],{"class":35,"line":119},[33,451,129],{"emptyLinePlaceholder":128},[33,453,454],{"class":35,"line":125},[33,455,456],{"class":149},"A:\n",[14,458,460],{"id":459},"extended-thinking-reasoning-mode","Extended Thinking \u002F Reasoning Mode",[19,462,464],{"filename":463,"language":22},"extended_thinking.py",[24,465,467],{"className":26,"code":466,"language":22,"meta":28,"style":28},"from anthropic import Anthropic\n\nclient = Anthropic()\n\n# Extended thinking is a DEDICATED reasoning phase with its own token budget,\n# distinct from prompted CoT. The model reasons in a separate channel BEFORE\n# producing its user-facing answer. The thinking content is typically presented\n# separately (collapsed\u002Fsummarized or not exposed at all), not interleaved.\n\nresponse = client.messages.create(\n    model=\"claude-opus-5\",\n    max_tokens=4096,\n    thinking={                  # ← dedicated reasoning budget, not prompt text\n        \"type\": \"enabled\",\n        \"budget_tokens\": 2048,  # the model can use up to 2048 tokens for reasoning\n    },\n    messages=[{\n        \"role\": \"user\",\n        \"content\": \"Given these three vendor contracts, which has the most \"\n                   \"unfavorable termination clause, and why?\",\n    }],\n)\n\n# The response contains separate blocks:\nthinking_block = next(b for b in response.content if b.type == \"thinking\")\nanswer_block = next(b for b in response.content if b.type == \"text\")\n\n# KEY DISTINCTION from prompted CoT:\n# - Prompted CoT: \"think step by step\" in the prompt → reasoning is VISIBLE\n#   in the response text, every time, even on trivial inputs\n# - Extended thinking: API-level setting → the model ADAPTIVELY decides how\n#   much reasoning a problem needs, reducing \"wasted CoT on easy tasks\"\n#\n# PRACTICAL RULE: on a model with genuine extended-thinking support, use the\n# dedicated setting\u002Fparameter rather than simulating it with a prompt instruction.\n# The dedicated mechanism is trained and optimized for deep reasoning.\n#\n# But: for tasks needing a SPECIFIC reasoning structure (Chapter 5's worked\n# examples, Chapter 11's verification checks), explicitly prompt that structure\n# — even alongside extended thinking. Extended thinking raises the CEILING on\n# unaided reasoning; it doesn't replace a scaffold you know works better.\n",[30,468,469,483,487,497,501,506,511,516,521,525,535,549,562,575,588,604,609,619,631,641,648,653,657,661,666,705,735,739,744,749,754,759,764,769,775,781,787,792,798,804,810],{"__ignoreMap":28},[33,470,471,474,477,480],{"class":35,"line":36},[33,472,473],{"class":141},"from",[33,475,476],{"class":149}," anthropic ",[33,478,479],{"class":141},"import",[33,481,482],{"class":149}," Anthropic\n",[33,484,485],{"class":35,"line":43},[33,486,129],{"emptyLinePlaceholder":128},[33,488,489,492,494],{"class":35,"line":49},[33,490,491],{"class":149},"client ",[33,493,165],{"class":141},[33,495,496],{"class":149}," Anthropic()\n",[33,498,499],{"class":35,"line":55},[33,500,129],{"emptyLinePlaceholder":128},[33,502,503],{"class":35,"line":61},[33,504,505],{"class":39},"# Extended thinking is a DEDICATED reasoning phase with its own token budget,\n",[33,507,508],{"class":35,"line":67},[33,509,510],{"class":39},"# distinct from prompted CoT. The model reasons in a separate channel BEFORE\n",[33,512,513],{"class":35,"line":73},[33,514,515],{"class":39},"# producing its user-facing answer. The thinking content is typically presented\n",[33,517,518],{"class":35,"line":78},[33,519,520],{"class":39},"# separately (collapsed\u002Fsummarized or not exposed at all), not interleaved.\n",[33,522,523],{"class":35,"line":84},[33,524,129],{"emptyLinePlaceholder":128},[33,526,527,530,532],{"class":35,"line":90},[33,528,529],{"class":149},"response ",[33,531,165],{"class":141},[33,533,534],{"class":149}," client.messages.create(\n",[33,536,537,541,543,546],{"class":35,"line":96},[33,538,540],{"class":539},"sCrzJ","    model",[33,542,165],{"class":141},[33,544,545],{"class":174},"\"claude-opus-5\"",[33,547,548],{"class":149},",\n",[33,550,551,554,556,560],{"class":35,"line":102},[33,552,553],{"class":539},"    max_tokens",[33,555,165],{"class":141},[33,557,559],{"class":558},"snvgF","4096",[33,561,548],{"class":149},[33,563,564,567,569,572],{"class":35,"line":107},[33,565,566],{"class":539},"    thinking",[33,568,165],{"class":141},[33,570,571],{"class":149},"{                  ",[33,573,574],{"class":39},"# ← dedicated reasoning budget, not prompt text\n",[33,576,577,580,583,586],{"class":35,"line":113},[33,578,579],{"class":174},"        \"type\"",[33,581,582],{"class":149},": ",[33,584,585],{"class":174},"\"enabled\"",[33,587,548],{"class":149},[33,589,590,593,595,598,601],{"class":35,"line":119},[33,591,592],{"class":174},"        \"budget_tokens\"",[33,594,582],{"class":149},[33,596,597],{"class":558},"2048",[33,599,600],{"class":149},",  ",[33,602,603],{"class":39},"# the model can use up to 2048 tokens for reasoning\n",[33,605,606],{"class":35,"line":125},[33,607,608],{"class":149},"    },\n",[33,610,611,614,616],{"class":35,"line":132},[33,612,613],{"class":539},"    messages",[33,615,165],{"class":141},[33,617,618],{"class":149},"[{\n",[33,620,621,624,626,629],{"class":35,"line":138},[33,622,623],{"class":174},"        \"role\"",[33,625,582],{"class":149},[33,627,628],{"class":174},"\"user\"",[33,630,548],{"class":149},[33,632,633,636,638],{"class":35,"line":153},[33,634,635],{"class":174},"        \"content\"",[33,637,582],{"class":149},[33,639,640],{"class":174},"\"Given these three vendor contracts, which has the most \"\n",[33,642,643,646],{"class":35,"line":159},[33,644,645],{"class":174},"                   \"unfavorable termination clause, and why?\"",[33,647,548],{"class":149},[33,649,650],{"class":35,"line":181},[33,651,652],{"class":149},"    }],\n",[33,654,655],{"class":35,"line":193},[33,656,178],{"class":149},[33,658,659],{"class":35,"line":198},[33,660,129],{"emptyLinePlaceholder":128},[33,662,663],{"class":35,"line":208},[33,664,665],{"class":39},"# The response contains separate blocks:\n",[33,667,668,671,673,676,679,682,685,688,691,694,697,700,703],{"class":35,"line":214},[33,669,670],{"class":149},"thinking_block ",[33,672,165],{"class":141},[33,674,675],{"class":558}," next",[33,677,678],{"class":149},"(b ",[33,680,681],{"class":141},"for",[33,683,684],{"class":149}," b ",[33,686,687],{"class":141},"in",[33,689,690],{"class":149}," response.content ",[33,692,693],{"class":141},"if",[33,695,696],{"class":149}," b.type ",[33,698,699],{"class":141},"==",[33,701,702],{"class":174}," \"thinking\"",[33,704,178],{"class":149},[33,706,707,710,712,714,716,718,720,722,724,726,728,730,733],{"class":35,"line":231},[33,708,709],{"class":149},"answer_block ",[33,711,165],{"class":141},[33,713,675],{"class":558},[33,715,678],{"class":149},[33,717,681],{"class":141},[33,719,684],{"class":149},[33,721,687],{"class":141},[33,723,690],{"class":149},[33,725,693],{"class":141},[33,727,696],{"class":149},[33,729,699],{"class":141},[33,731,732],{"class":174}," \"text\"",[33,734,178],{"class":149},[33,736,737],{"class":35,"line":251},[33,738,129],{"emptyLinePlaceholder":128},[33,740,741],{"class":35,"line":261},[33,742,743],{"class":39},"# KEY DISTINCTION from prompted CoT:\n",[33,745,746],{"class":35,"line":266},[33,747,748],{"class":39},"# - Prompted CoT: \"think step by step\" in the prompt → reasoning is VISIBLE\n",[33,750,751],{"class":35,"line":272},[33,752,753],{"class":39},"#   in the response text, every time, even on trivial inputs\n",[33,755,756],{"class":35,"line":278},[33,757,758],{"class":39},"# - Extended thinking: API-level setting → the model ADAPTIVELY decides how\n",[33,760,761],{"class":35,"line":284},[33,762,763],{"class":39},"#   much reasoning a problem needs, reducing \"wasted CoT on easy tasks\"\n",[33,765,767],{"class":35,"line":766},33,[33,768,46],{"class":39},[33,770,772],{"class":35,"line":771},34,[33,773,774],{"class":39},"# PRACTICAL RULE: on a model with genuine extended-thinking support, use the\n",[33,776,778],{"class":35,"line":777},35,[33,779,780],{"class":39},"# dedicated setting\u002Fparameter rather than simulating it with a prompt instruction.\n",[33,782,784],{"class":35,"line":783},36,[33,785,786],{"class":39},"# The dedicated mechanism is trained and optimized for deep reasoning.\n",[33,788,790],{"class":35,"line":789},37,[33,791,46],{"class":39},[33,793,795],{"class":35,"line":794},38,[33,796,797],{"class":39},"# But: for tasks needing a SPECIFIC reasoning structure (Chapter 5's worked\n",[33,799,801],{"class":35,"line":800},39,[33,802,803],{"class":39},"# examples, Chapter 11's verification checks), explicitly prompt that structure\n",[33,805,807],{"class":35,"line":806},40,[33,808,809],{"class":39},"# — even alongside extended thinking. Extended thinking raises the CEILING on\n",[33,811,813],{"class":35,"line":812},41,[33,814,815],{"class":39},"# unaided reasoning; it doesn't replace a scaffold you know works better.\n",[14,817,819],{"id":818},"anti-pattern-cot-on-simple-tasks","Anti-Pattern: CoT on Simple Tasks",[19,821,823],{"filename":822,"language":22},"anti_patterns.py",[24,824,826],{"className":26,"code":825,"language":22,"meta":28,"style":28},"# ANTI-PATTERN: reflexively adding \"think step by step\" to EVERY prompt\n\n# BAD — CoT on a simple classification adds latency + cost for zero benefit:\nBAD_PROMPT = \"\"\"\nClassify the sentiment of this review as POSITIVE, NEGATIVE, or MIXED.\n\nReview: \"Works great, arrived on time.\"\n\nThink through this step by step before giving your final answer.\n\"\"\"\n# The model might overthink a simple case into an incorrect, more nuanced-sounding\n# but wrong answer — talking itself out of the obviously correct response.\n\n# FINE — direct answer on a simple task:\nGOOD_PROMPT = \"\"\"\nClassify the sentiment of this review as POSITIVE, NEGATIVE, or MIXED.\nReply with only the label.\n\nReview: \"Works great, arrived on time.\"\n\"\"\"\n\n# ANTI-PATTERN: reasoning AFTER the answer\nBAD_ORDERING = \"\"\"\nWhat is the final price? Answer first, then explain your reasoning.\n\"\"\"\n# Because generation is autoregressive, reasoning that appears AFTER a stated\n# answer CANNOT have influenced that answer — the answer was already committed\n# before the reasoning tokens were generated. This is post-hoc justification,\n# not reasoning that informs the answer.\n\n# CORRECT: reasoning BEFORE the answer\nGOOD_ORDERING = \"\"\"\nThink through this step by step, then give your final answer.\n\"\"\"\n# The reasoning tokens become context that the answer token conditions on.\n",[30,827,828,833,837,842,853,858,862,867,871,875,880,885,890,894,899,908,912,917,921,925,929,933,938,947,952,956,961,966,971,976,980,985,994,999,1003],{"__ignoreMap":28},[33,829,830],{"class":35,"line":36},[33,831,832],{"class":39},"# ANTI-PATTERN: reflexively adding \"think step by step\" to EVERY prompt\n",[33,834,835],{"class":35,"line":43},[33,836,129],{"emptyLinePlaceholder":128},[33,838,839],{"class":35,"line":49},[33,840,841],{"class":39},"# BAD — CoT on a simple classification adds latency + cost for zero benefit:\n",[33,843,844,847,850],{"class":35,"line":55},[33,845,846],{"class":558},"BAD_PROMPT",[33,848,849],{"class":141}," =",[33,851,852],{"class":174}," \"\"\"\n",[33,854,855],{"class":35,"line":61},[33,856,857],{"class":174},"Classify the sentiment of this review as POSITIVE, NEGATIVE, or MIXED.\n",[33,859,860],{"class":35,"line":67},[33,861,129],{"emptyLinePlaceholder":128},[33,863,864],{"class":35,"line":73},[33,865,866],{"class":174},"Review: \"Works great, arrived on time.\"\n",[33,868,869],{"class":35,"line":78},[33,870,129],{"emptyLinePlaceholder":128},[33,872,873],{"class":35,"line":84},[33,874,325],{"class":174},[33,876,877],{"class":35,"line":90},[33,878,879],{"class":174},"\"\"\"\n",[33,881,882],{"class":35,"line":96},[33,883,884],{"class":39},"# The model might overthink a simple case into an incorrect, more nuanced-sounding\n",[33,886,887],{"class":35,"line":102},[33,888,889],{"class":39},"# but wrong answer — talking itself out of the obviously correct response.\n",[33,891,892],{"class":35,"line":107},[33,893,129],{"emptyLinePlaceholder":128},[33,895,896],{"class":35,"line":113},[33,897,898],{"class":39},"# FINE — direct answer on a simple task:\n",[33,900,901,904,906],{"class":35,"line":119},[33,902,903],{"class":558},"GOOD_PROMPT",[33,905,849],{"class":141},[33,907,852],{"class":174},[33,909,910],{"class":35,"line":125},[33,911,857],{"class":174},[33,913,914],{"class":35,"line":132},[33,915,916],{"class":174},"Reply with only the label.\n",[33,918,919],{"class":35,"line":138},[33,920,129],{"emptyLinePlaceholder":128},[33,922,923],{"class":35,"line":153},[33,924,866],{"class":174},[33,926,927],{"class":35,"line":159},[33,928,879],{"class":174},[33,930,931],{"class":35,"line":181},[33,932,129],{"emptyLinePlaceholder":128},[33,934,935],{"class":35,"line":193},[33,936,937],{"class":39},"# ANTI-PATTERN: reasoning AFTER the answer\n",[33,939,940,943,945],{"class":35,"line":198},[33,941,942],{"class":558},"BAD_ORDERING",[33,944,849],{"class":141},[33,946,852],{"class":174},[33,948,949],{"class":35,"line":208},[33,950,951],{"class":174},"What is the final price? Answer first, then explain your reasoning.\n",[33,953,954],{"class":35,"line":214},[33,955,879],{"class":174},[33,957,958],{"class":35,"line":231},[33,959,960],{"class":39},"# Because generation is autoregressive, reasoning that appears AFTER a stated\n",[33,962,963],{"class":35,"line":251},[33,964,965],{"class":39},"# answer CANNOT have influenced that answer — the answer was already committed\n",[33,967,968],{"class":35,"line":261},[33,969,970],{"class":39},"# before the reasoning tokens were generated. This is post-hoc justification,\n",[33,972,973],{"class":35,"line":266},[33,974,975],{"class":39},"# not reasoning that informs the answer.\n",[33,977,978],{"class":35,"line":272},[33,979,129],{"emptyLinePlaceholder":128},[33,981,982],{"class":35,"line":278},[33,983,984],{"class":39},"# CORRECT: reasoning BEFORE the answer\n",[33,986,987,990,992],{"class":35,"line":284},[33,988,989],{"class":558},"GOOD_ORDERING",[33,991,849],{"class":141},[33,993,852],{"class":174},[33,995,996],{"class":35,"line":766},[33,997,998],{"class":174},"Think through this step by step, then give your final answer.\n",[33,1000,1001],{"class":35,"line":771},[33,1002,879],{"class":174},[33,1004,1005],{"class":35,"line":777},[33,1006,1007],{"class":39},"# The reasoning tokens become context that the answer token conditions on.\n",[14,1009,1011],{"id":1010},"cot-output-format-the-separation-pattern","CoT + Output Format: The Separation Pattern",[19,1013,1015],{"filename":1014,"language":295},"cot_with_format.md",[24,1016,1018],{"className":298,"code":1017,"language":295,"meta":28,"style":28},"First, reason through the problem step by step inside \u003Creasoning> tags.\nThen, after your reasoning, output your final answer inside \u003Canswer> tags\nas a JSON object matching this schema: {\"category\": string, \"confidence\":\n\"low\" | \"medium\" | \"high\"}. Nothing should appear after the closing\n\u003C\u002Fanswer> tag.\n",[30,1019,1020,1025,1030,1035,1040],{"__ignoreMap":28},[33,1021,1022],{"class":35,"line":36},[33,1023,1024],{"class":149},"First, reason through the problem step by step inside \u003Creasoning> tags.\n",[33,1026,1027],{"class":35,"line":43},[33,1028,1029],{"class":149},"Then, after your reasoning, output your final answer inside \u003Canswer> tags\n",[33,1031,1032],{"class":35,"line":49},[33,1033,1034],{"class":149},"as a JSON object matching this schema: {\"category\": string, \"confidence\":\n",[33,1036,1037],{"class":35,"line":55},[33,1038,1039],{"class":149},"\"low\" | \"medium\" | \"high\"}. Nothing should appear after the closing\n",[33,1041,1042],{"class":35,"line":61},[33,1043,1044],{"class":149},"\u003C\u002Fanswer> tag.\n",[19,1046,1048],{"filename":1047,"language":22},"two_call_pipeline.py",[24,1049,1051],{"className":26,"code":1050,"language":22,"meta":28,"style":28},"import re, json\n\n# PRODUCTION PATTERN: two-call pipeline cleanly separates reasoning from\n# structured output, avoiding the format conflict entirely.\n\ndef reasoning_then_extraction(question: str, context: str) -> dict:\n    # Call 1: reason freely (no format constraint competing for attention)\n    reasoning_response = client.messages.create(\n        model=\"claude-opus-5\",\n        max_tokens=2048,\n        messages=[{\"role\": \"user\", \"content\": f\"\"\"\n            Reason through this problem step by step.\n            Context: {context}\n            Question: {question}\n            Show your work. Do not produce a final structured answer yet.\n        \"\"\"}],\n    )\n    reasoning = reasoning_response.content[0].text\n\n    # Call 2: extract structured answer from the reasoning (format-only, no reasoning)\n    extraction_response = client.messages.create(\n        model=\"claude-opus-5\",\n        max_tokens=256,\n        output_config={\"format\": {\"type\": \"json_schema\", \"schema\": {\n            \"type\": \"object\",\n            \"properties\": {\n                \"category\": {\"type\": \"string\"},\n                \"confidence\": {\"type\": \"string\", \"enum\": [\"low\", \"medium\", \"high\"]},\n            },\n            \"required\": [\"category\", \"confidence\"],\n        }}},\n        messages=[{\"role\": \"user\", \"content\": f\"\"\"\n            Based on this reasoning, extract the final answer as JSON.\n            Reasoning: {reasoning}\n        \"\"\"}],\n    )\n    return json.loads(extraction_response.content[0].text)\n\n# Benefits: reasoning is free-form (no format tension), structured output is\n# enforced by the API (no parsing failures), and you can LOG the reasoning\n# separately for debugging without it polluting the parsed result.\n",[30,1052,1053,1060,1064,1069,1074,1078,1105,1110,1119,1130,1141,1171,1176,1190,1202,1207,1215,1220,1235,1239,1244,1253,1263,1274,1305,1317,1324,1341,1378,1383,1401,1406,1430,1435,1447,1453,1457,1469,1473,1478,1483],{"__ignoreMap":28},[33,1054,1055,1057],{"class":35,"line":36},[33,1056,479],{"class":141},[33,1058,1059],{"class":149}," re, json\n",[33,1061,1062],{"class":35,"line":43},[33,1063,129],{"emptyLinePlaceholder":128},[33,1065,1066],{"class":35,"line":49},[33,1067,1068],{"class":39},"# PRODUCTION PATTERN: two-call pipeline cleanly separates reasoning from\n",[33,1070,1071],{"class":35,"line":55},[33,1072,1073],{"class":39},"# structured output, avoiding the format conflict entirely.\n",[33,1075,1076],{"class":35,"line":61},[33,1077,129],{"emptyLinePlaceholder":128},[33,1079,1080,1082,1085,1088,1091,1094,1096,1099,1102],{"class":35,"line":67},[33,1081,142],{"class":141},[33,1083,1084],{"class":145}," reasoning_then_extraction",[33,1086,1087],{"class":149},"(question: ",[33,1089,1090],{"class":558},"str",[33,1092,1093],{"class":149},", context: ",[33,1095,1090],{"class":558},[33,1097,1098],{"class":149},") -> ",[33,1100,1101],{"class":558},"dict",[33,1103,1104],{"class":149},":\n",[33,1106,1107],{"class":35,"line":73},[33,1108,1109],{"class":39},"    # Call 1: reason freely (no format constraint competing for attention)\n",[33,1111,1112,1115,1117],{"class":35,"line":78},[33,1113,1114],{"class":149},"    reasoning_response ",[33,1116,165],{"class":141},[33,1118,534],{"class":149},[33,1120,1121,1124,1126,1128],{"class":35,"line":84},[33,1122,1123],{"class":539},"        model",[33,1125,165],{"class":141},[33,1127,545],{"class":174},[33,1129,548],{"class":149},[33,1131,1132,1135,1137,1139],{"class":35,"line":90},[33,1133,1134],{"class":539},"        max_tokens",[33,1136,165],{"class":141},[33,1138,597],{"class":558},[33,1140,548],{"class":149},[33,1142,1143,1146,1148,1151,1154,1156,1158,1161,1164,1166,1169],{"class":35,"line":96},[33,1144,1145],{"class":539},"        messages",[33,1147,165],{"class":141},[33,1149,1150],{"class":149},"[{",[33,1152,1153],{"class":174},"\"role\"",[33,1155,582],{"class":149},[33,1157,628],{"class":174},[33,1159,1160],{"class":149},", ",[33,1162,1163],{"class":174},"\"content\"",[33,1165,582],{"class":149},[33,1167,1168],{"class":141},"f",[33,1170,879],{"class":174},[33,1172,1173],{"class":35,"line":102},[33,1174,1175],{"class":174},"            Reason through this problem step by step.\n",[33,1177,1178,1181,1184,1187],{"class":35,"line":107},[33,1179,1180],{"class":174},"            Context: ",[33,1182,1183],{"class":558},"{",[33,1185,1186],{"class":149},"context",[33,1188,1189],{"class":558},"}\n",[33,1191,1192,1195,1197,1200],{"class":35,"line":113},[33,1193,1194],{"class":174},"            Question: ",[33,1196,1183],{"class":558},[33,1198,1199],{"class":149},"question",[33,1201,1189],{"class":558},[33,1203,1204],{"class":35,"line":119},[33,1205,1206],{"class":174},"            Show your work. Do not produce a final structured answer yet.\n",[33,1208,1209,1212],{"class":35,"line":125},[33,1210,1211],{"class":174},"        \"\"\"",[33,1213,1214],{"class":149},"}],\n",[33,1216,1217],{"class":35,"line":132},[33,1218,1219],{"class":149},"    )\n",[33,1221,1222,1224,1226,1229,1232],{"class":35,"line":138},[33,1223,217],{"class":149},[33,1225,165],{"class":141},[33,1227,1228],{"class":149}," reasoning_response.content[",[33,1230,1231],{"class":558},"0",[33,1233,1234],{"class":149},"].text\n",[33,1236,1237],{"class":35,"line":153},[33,1238,129],{"emptyLinePlaceholder":128},[33,1240,1241],{"class":35,"line":159},[33,1242,1243],{"class":39},"    # Call 2: extract structured answer from the reasoning (format-only, no reasoning)\n",[33,1245,1246,1249,1251],{"class":35,"line":181},[33,1247,1248],{"class":149},"    extraction_response ",[33,1250,165],{"class":141},[33,1252,534],{"class":149},[33,1254,1255,1257,1259,1261],{"class":35,"line":193},[33,1256,1123],{"class":539},[33,1258,165],{"class":141},[33,1260,545],{"class":174},[33,1262,548],{"class":149},[33,1264,1265,1267,1269,1272],{"class":35,"line":198},[33,1266,1134],{"class":539},[33,1268,165],{"class":141},[33,1270,1271],{"class":558},"256",[33,1273,548],{"class":149},[33,1275,1276,1279,1281,1283,1286,1289,1292,1294,1297,1299,1302],{"class":35,"line":208},[33,1277,1278],{"class":539},"        output_config",[33,1280,165],{"class":141},[33,1282,1183],{"class":149},[33,1284,1285],{"class":174},"\"format\"",[33,1287,1288],{"class":149},": {",[33,1290,1291],{"class":174},"\"type\"",[33,1293,582],{"class":149},[33,1295,1296],{"class":174},"\"json_schema\"",[33,1298,1160],{"class":149},[33,1300,1301],{"class":174},"\"schema\"",[33,1303,1304],{"class":149},": {\n",[33,1306,1307,1310,1312,1315],{"class":35,"line":214},[33,1308,1309],{"class":174},"            \"type\"",[33,1311,582],{"class":149},[33,1313,1314],{"class":174},"\"object\"",[33,1316,548],{"class":149},[33,1318,1319,1322],{"class":35,"line":231},[33,1320,1321],{"class":174},"            \"properties\"",[33,1323,1304],{"class":149},[33,1325,1326,1329,1331,1333,1335,1338],{"class":35,"line":251},[33,1327,1328],{"class":174},"                \"category\"",[33,1330,1288],{"class":149},[33,1332,1291],{"class":174},[33,1334,582],{"class":149},[33,1336,1337],{"class":174},"\"string\"",[33,1339,1340],{"class":149},"},\n",[33,1342,1343,1346,1348,1350,1352,1354,1356,1359,1362,1365,1367,1370,1372,1375],{"class":35,"line":261},[33,1344,1345],{"class":174},"                \"confidence\"",[33,1347,1288],{"class":149},[33,1349,1291],{"class":174},[33,1351,582],{"class":149},[33,1353,1337],{"class":174},[33,1355,1160],{"class":149},[33,1357,1358],{"class":174},"\"enum\"",[33,1360,1361],{"class":149},": [",[33,1363,1364],{"class":174},"\"low\"",[33,1366,1160],{"class":149},[33,1368,1369],{"class":174},"\"medium\"",[33,1371,1160],{"class":149},[33,1373,1374],{"class":174},"\"high\"",[33,1376,1377],{"class":149},"]},\n",[33,1379,1380],{"class":35,"line":266},[33,1381,1382],{"class":149},"            },\n",[33,1384,1385,1388,1390,1393,1395,1398],{"class":35,"line":272},[33,1386,1387],{"class":174},"            \"required\"",[33,1389,1361],{"class":149},[33,1391,1392],{"class":174},"\"category\"",[33,1394,1160],{"class":149},[33,1396,1397],{"class":174},"\"confidence\"",[33,1399,1400],{"class":149},"],\n",[33,1402,1403],{"class":35,"line":278},[33,1404,1405],{"class":149},"        }}},\n",[33,1407,1408,1410,1412,1414,1416,1418,1420,1422,1424,1426,1428],{"class":35,"line":284},[33,1409,1145],{"class":539},[33,1411,165],{"class":141},[33,1413,1150],{"class":149},[33,1415,1153],{"class":174},[33,1417,582],{"class":149},[33,1419,628],{"class":174},[33,1421,1160],{"class":149},[33,1423,1163],{"class":174},[33,1425,582],{"class":149},[33,1427,1168],{"class":141},[33,1429,879],{"class":174},[33,1431,1432],{"class":35,"line":766},[33,1433,1434],{"class":174},"            Based on this reasoning, extract the final answer as JSON.\n",[33,1436,1437,1440,1442,1445],{"class":35,"line":771},[33,1438,1439],{"class":174},"            Reasoning: ",[33,1441,1183],{"class":558},[33,1443,1444],{"class":149},"reasoning",[33,1446,1189],{"class":558},[33,1448,1449,1451],{"class":35,"line":777},[33,1450,1211],{"class":174},[33,1452,1214],{"class":149},[33,1454,1455],{"class":35,"line":783},[33,1456,1219],{"class":149},[33,1458,1459,1461,1464,1466],{"class":35,"line":789},[33,1460,184],{"class":141},[33,1462,1463],{"class":149}," json.loads(extraction_response.content[",[33,1465,1231],{"class":558},[33,1467,1468],{"class":149},"].text)\n",[33,1470,1471],{"class":35,"line":794},[33,1472,129],{"emptyLinePlaceholder":128},[33,1474,1475],{"class":35,"line":800},[33,1476,1477],{"class":39},"# Benefits: reasoning is free-form (no format tension), structured output is\n",[33,1479,1480],{"class":35,"line":806},[33,1481,1482],{"class":39},"# enforced by the API (no parsing failures), and you can LOG the reasoning\n",[33,1484,1485],{"class":35,"line":812},[33,1486,1487],{"class":39},"# separately for debugging without it polluting the parsed result.\n",[14,1489,1491],{"id":1490},"the-fluent-but-wrong-failure","The \"Fluent But Wrong\" Failure",[19,1493,1495],{"filename":1494,"language":295},"fluent_but_wrong.md",[24,1496,1498],{"className":298,"code":1497,"language":295,"meta":28,"style":28},"Determine whether this applicant qualifies for pre-approval. Think step\nby step, and make sure your final answer is APPROVED or DENIED.\n\nApplicant: credit score 710, annual income $58,000, requested loan\namount $340,000, existing monthly debt payments $1,200.\n\nApproval requires: credit score >= 680, debt-to-income ratio (existing\nmonthly debt \u002F monthly income) below 36%, and loan amount no more than\n5x annual income.\n",[30,1499,1500,1505,1510,1514,1519,1524,1528,1533,1538],{"__ignoreMap":28},[33,1501,1502],{"class":35,"line":36},[33,1503,1504],{"class":149},"Determine whether this applicant qualifies for pre-approval. Think step\n",[33,1506,1507],{"class":35,"line":43},[33,1508,1509],{"class":149},"by step, and make sure your final answer is APPROVED or DENIED.\n",[33,1511,1512],{"class":35,"line":49},[33,1513,129],{"emptyLinePlaceholder":128},[33,1515,1516],{"class":35,"line":55},[33,1517,1518],{"class":149},"Applicant: credit score 710, annual income $58,000, requested loan\n",[33,1520,1521],{"class":35,"line":61},[33,1522,1523],{"class":149},"amount $340,000, existing monthly debt payments $1,200.\n",[33,1525,1526],{"class":35,"line":67},[33,1527,129],{"emptyLinePlaceholder":128},[33,1529,1530],{"class":35,"line":73},[33,1531,1532],{"class":149},"Approval requires: credit score >= 680, debt-to-income ratio (existing\n",[33,1534,1535],{"class":35,"line":78},[33,1536,1537],{"class":149},"monthly debt \u002F monthly income) below 36%, and loan amount no more than\n",[33,1539,1540],{"class":35,"line":84},[33,1541,1542],{"class":149},"5x annual income.\n",[19,1544,1546],{"filename":1545,"language":22},"fluent_but_wrong_diagnosis.py",[24,1547,1549],{"className":26,"code":1548,"language":22,"meta":28,"style":28},"# The model produces a lengthy, confident chain of reasoning and concludes APPROVED.\n# A loan officer catches: $340,000 \u002F $58,000 = 5.86x — FAILS the \"no more than 5x\" rule.\n# Yet the model's reasoning trace CLAIMED to check this criterion and stated it passed.\n#\n# WHAT HAPPENED: the reasoning trace LOOKED like it verified the constraint, but\n# the underlying multiplication (5 × $58,000 = $290,000, then $340K vs $290K) either\n# wasn't performed correctly or was misreported in the restated conclusion.\n#\n# THE DEEPER LESSON: a visible reasoning trace is NOT a verification mechanism.\n# It's a debugging aid and an accuracy improvement ON AVERAGE, but any individual\n# trace can be wrong while looking entirely legitimate. Fluency ≠ correctness.\n#\n# PRODUCTION FIX for high-stakes numeric\u002Frule-based decisions:\n# Externalize the actual arithmetic and threshold checks into DETERMINISTIC code.\n\ndef evaluate_loan(credit_score, annual_income, loan_amount, monthly_debt):\n    \"\"\"Deterministic evaluation — the model does NOT compute these.\"\"\"\n    # Rule 1: credit score >= 680\n    credit_ok = credit_score >= 680\n\n    # Rule 2: DTI \u003C 36% (existing monthly debt \u002F monthly income)\n    monthly_income = annual_income \u002F 12\n    dti = monthly_debt \u002F monthly_income\n    dti_ok = dti \u003C 0.36\n\n    # Rule 3: loan amount \u003C= 5x annual income\n    loan_ratio_ok = loan_amount \u003C= 5 * annual_income\n\n    return {\n        \"approved\": credit_ok and dti_ok and loan_ratio_ok,\n        \"checks\": {\n            \"credit_score\": {\"value\": credit_score, \"pass\": credit_ok},\n            \"dti_ratio\": {\"value\": round(dti, 4), \"pass\": dti_ok},\n            \"loan_to_income\": {\"value\": round(loan_amount \u002F annual_income, 2), \"pass\": loan_ratio_ok},\n        }\n    }\n\nresult = evaluate_loan(710, 58_000, 340_000, 1_200)\n# {\"approved\": False, \"checks\": {\n#   \"credit_score\": {\"value\": 710, \"pass\": True},\n#   \"dti_ratio\": {\"value\": 0.0248, \"pass\": True},\n#   \"loan_to_income\": {\"value\": 5.86, \"pass\": False}  ← deterministic, no hallucination\n# }}\n\n# Use the MODEL for judgment (is the application suspicious? is there context?),\n# use DETERMINISTIC CODE for the numbers. CoT is a nice-to-have explanation\n# layer on top of verified numbers, NOT the source of truth for the numbers.\n",[30,1550,1551,1556,1561,1566,1570,1575,1580,1585,1589,1594,1599,1604,1608,1613,1618,1622,1632,1637,1642,1658,1662,1667,1683,1698,1714,1718,1723,1745,1749,1756,1775,1782,1801,1829,1860,1865,1870,1874,1904,1909,1914,1919,1925,1931,1936,1942,1948],{"__ignoreMap":28},[33,1552,1553],{"class":35,"line":36},[33,1554,1555],{"class":39},"# The model produces a lengthy, confident chain of reasoning and concludes APPROVED.\n",[33,1557,1558],{"class":35,"line":43},[33,1559,1560],{"class":39},"# A loan officer catches: $340,000 \u002F $58,000 = 5.86x — FAILS the \"no more than 5x\" rule.\n",[33,1562,1563],{"class":35,"line":49},[33,1564,1565],{"class":39},"# Yet the model's reasoning trace CLAIMED to check this criterion and stated it passed.\n",[33,1567,1568],{"class":35,"line":55},[33,1569,46],{"class":39},[33,1571,1572],{"class":35,"line":61},[33,1573,1574],{"class":39},"# WHAT HAPPENED: the reasoning trace LOOKED like it verified the constraint, but\n",[33,1576,1577],{"class":35,"line":67},[33,1578,1579],{"class":39},"# the underlying multiplication (5 × $58,000 = $290,000, then $340K vs $290K) either\n",[33,1581,1582],{"class":35,"line":73},[33,1583,1584],{"class":39},"# wasn't performed correctly or was misreported in the restated conclusion.\n",[33,1586,1587],{"class":35,"line":78},[33,1588,46],{"class":39},[33,1590,1591],{"class":35,"line":84},[33,1592,1593],{"class":39},"# THE DEEPER LESSON: a visible reasoning trace is NOT a verification mechanism.\n",[33,1595,1596],{"class":35,"line":90},[33,1597,1598],{"class":39},"# It's a debugging aid and an accuracy improvement ON AVERAGE, but any individual\n",[33,1600,1601],{"class":35,"line":96},[33,1602,1603],{"class":39},"# trace can be wrong while looking entirely legitimate. Fluency ≠ correctness.\n",[33,1605,1606],{"class":35,"line":102},[33,1607,46],{"class":39},[33,1609,1610],{"class":35,"line":107},[33,1611,1612],{"class":39},"# PRODUCTION FIX for high-stakes numeric\u002Frule-based decisions:\n",[33,1614,1615],{"class":35,"line":113},[33,1616,1617],{"class":39},"# Externalize the actual arithmetic and threshold checks into DETERMINISTIC code.\n",[33,1619,1620],{"class":35,"line":119},[33,1621,129],{"emptyLinePlaceholder":128},[33,1623,1624,1626,1629],{"class":35,"line":125},[33,1625,142],{"class":141},[33,1627,1628],{"class":145}," evaluate_loan",[33,1630,1631],{"class":149},"(credit_score, annual_income, loan_amount, monthly_debt):\n",[33,1633,1634],{"class":35,"line":132},[33,1635,1636],{"class":174},"    \"\"\"Deterministic evaluation — the model does NOT compute these.\"\"\"\n",[33,1638,1639],{"class":35,"line":138},[33,1640,1641],{"class":39},"    # Rule 1: credit score >= 680\n",[33,1643,1644,1647,1649,1652,1655],{"class":35,"line":153},[33,1645,1646],{"class":149},"    credit_ok ",[33,1648,165],{"class":141},[33,1650,1651],{"class":149}," credit_score ",[33,1653,1654],{"class":141},">=",[33,1656,1657],{"class":558}," 680\n",[33,1659,1660],{"class":35,"line":159},[33,1661,129],{"emptyLinePlaceholder":128},[33,1663,1664],{"class":35,"line":181},[33,1665,1666],{"class":39},"    # Rule 2: DTI \u003C 36% (existing monthly debt \u002F monthly income)\n",[33,1668,1669,1672,1674,1677,1680],{"class":35,"line":193},[33,1670,1671],{"class":149},"    monthly_income ",[33,1673,165],{"class":141},[33,1675,1676],{"class":149}," annual_income ",[33,1678,1679],{"class":141},"\u002F",[33,1681,1682],{"class":558}," 12\n",[33,1684,1685,1688,1690,1693,1695],{"class":35,"line":198},[33,1686,1687],{"class":149},"    dti ",[33,1689,165],{"class":141},[33,1691,1692],{"class":149}," monthly_debt ",[33,1694,1679],{"class":141},[33,1696,1697],{"class":149}," monthly_income\n",[33,1699,1700,1703,1705,1708,1711],{"class":35,"line":208},[33,1701,1702],{"class":149},"    dti_ok ",[33,1704,165],{"class":141},[33,1706,1707],{"class":149}," dti ",[33,1709,1710],{"class":141},"\u003C",[33,1712,1713],{"class":558}," 0.36\n",[33,1715,1716],{"class":35,"line":214},[33,1717,129],{"emptyLinePlaceholder":128},[33,1719,1720],{"class":35,"line":231},[33,1721,1722],{"class":39},"    # Rule 3: loan amount \u003C= 5x annual income\n",[33,1724,1725,1728,1730,1733,1736,1739,1742],{"class":35,"line":251},[33,1726,1727],{"class":149},"    loan_ratio_ok ",[33,1729,165],{"class":141},[33,1731,1732],{"class":149}," loan_amount ",[33,1734,1735],{"class":141},"\u003C=",[33,1737,1738],{"class":558}," 5",[33,1740,1741],{"class":141}," *",[33,1743,1744],{"class":149}," annual_income\n",[33,1746,1747],{"class":35,"line":261},[33,1748,129],{"emptyLinePlaceholder":128},[33,1750,1751,1753],{"class":35,"line":266},[33,1752,184],{"class":141},[33,1754,1755],{"class":149}," {\n",[33,1757,1758,1761,1764,1767,1770,1772],{"class":35,"line":272},[33,1759,1760],{"class":174},"        \"approved\"",[33,1762,1763],{"class":149},": credit_ok ",[33,1765,1766],{"class":141},"and",[33,1768,1769],{"class":149}," dti_ok ",[33,1771,1766],{"class":141},[33,1773,1774],{"class":149}," loan_ratio_ok,\n",[33,1776,1777,1780],{"class":35,"line":278},[33,1778,1779],{"class":174},"        \"checks\"",[33,1781,1304],{"class":149},[33,1783,1784,1787,1789,1792,1795,1798],{"class":35,"line":284},[33,1785,1786],{"class":174},"            \"credit_score\"",[33,1788,1288],{"class":149},[33,1790,1791],{"class":174},"\"value\"",[33,1793,1794],{"class":149},": credit_score, ",[33,1796,1797],{"class":174},"\"pass\"",[33,1799,1800],{"class":149},": credit_ok},\n",[33,1802,1803,1806,1808,1810,1812,1815,1818,1821,1824,1826],{"class":35,"line":766},[33,1804,1805],{"class":174},"            \"dti_ratio\"",[33,1807,1288],{"class":149},[33,1809,1791],{"class":174},[33,1811,582],{"class":149},[33,1813,1814],{"class":558},"round",[33,1816,1817],{"class":149},"(dti, ",[33,1819,1820],{"class":558},"4",[33,1822,1823],{"class":149},"), ",[33,1825,1797],{"class":174},[33,1827,1828],{"class":149},": dti_ok},\n",[33,1830,1831,1834,1836,1838,1840,1842,1845,1847,1850,1853,1855,1857],{"class":35,"line":771},[33,1832,1833],{"class":174},"            \"loan_to_income\"",[33,1835,1288],{"class":149},[33,1837,1791],{"class":174},[33,1839,582],{"class":149},[33,1841,1814],{"class":558},[33,1843,1844],{"class":149},"(loan_amount ",[33,1846,1679],{"class":141},[33,1848,1849],{"class":149}," annual_income, ",[33,1851,1852],{"class":558},"2",[33,1854,1823],{"class":149},[33,1856,1797],{"class":174},[33,1858,1859],{"class":149},": loan_ratio_ok},\n",[33,1861,1862],{"class":35,"line":777},[33,1863,1864],{"class":149},"        }\n",[33,1866,1867],{"class":35,"line":783},[33,1868,1869],{"class":149},"    }\n",[33,1871,1872],{"class":35,"line":789},[33,1873,129],{"emptyLinePlaceholder":128},[33,1875,1876,1879,1881,1884,1887,1889,1892,1894,1897,1899,1902],{"class":35,"line":794},[33,1877,1878],{"class":149},"result ",[33,1880,165],{"class":141},[33,1882,1883],{"class":149}," evaluate_loan(",[33,1885,1886],{"class":558},"710",[33,1888,1160],{"class":149},[33,1890,1891],{"class":558},"58_000",[33,1893,1160],{"class":149},[33,1895,1896],{"class":558},"340_000",[33,1898,1160],{"class":149},[33,1900,1901],{"class":558},"1_200",[33,1903,178],{"class":149},[33,1905,1906],{"class":35,"line":800},[33,1907,1908],{"class":39},"# {\"approved\": False, \"checks\": {\n",[33,1910,1911],{"class":35,"line":806},[33,1912,1913],{"class":39},"#   \"credit_score\": {\"value\": 710, \"pass\": True},\n",[33,1915,1916],{"class":35,"line":812},[33,1917,1918],{"class":39},"#   \"dti_ratio\": {\"value\": 0.0248, \"pass\": True},\n",[33,1920,1922],{"class":35,"line":1921},42,[33,1923,1924],{"class":39},"#   \"loan_to_income\": {\"value\": 5.86, \"pass\": False}  ← deterministic, no hallucination\n",[33,1926,1928],{"class":35,"line":1927},43,[33,1929,1930],{"class":39},"# }}\n",[33,1932,1934],{"class":35,"line":1933},44,[33,1935,129],{"emptyLinePlaceholder":128},[33,1937,1939],{"class":35,"line":1938},45,[33,1940,1941],{"class":39},"# Use the MODEL for judgment (is the application suspicious? is there context?),\n",[33,1943,1945],{"class":35,"line":1944},46,[33,1946,1947],{"class":39},"# use DETERMINISTIC CODE for the numbers. CoT is a nice-to-have explanation\n",[33,1949,1951],{"class":35,"line":1950},47,[33,1952,1953],{"class":39},"# layer on top of verified numbers, NOT the source of truth for the numbers.\n",[14,1955,1957],{"id":1956},"tips-tricks","💡 Tips & Tricks",[19,1959,1961],{"filename":1960,"language":22},"tips.py",[24,1962,1964],{"className":26,"code":1963,"language":22,"meta":28,"style":28},"# [Idiom] \"Think step by step\" is a floor, not a ceiling. Specify WHAT KIND\n# of steps: \"list the relevant constraints first, then check each option\n# against them, then state your conclusion\" gives a specific reasoning scaffold.\n\n# [Debug] Ask for reasoning BEFORE the answer, never after. Autoregressive\n# generation means reasoning after the answer is post-hoc justification that\n# couldn't have influenced the answer.\n\n# [Debug] Use CoT to debug prompts even if you don't ship it. When a zero-shot\n# prompt fails mysteriously, add \"think step by step\" and inspect the trace.\n# It often reveals which assumption or ambiguity is causing the failure (Chapter 4).\n# Then fix the instruction — sometimes letting you remove the CoT entirely.\n\n# [Performance] Cap reasoning length for cost-sensitive paths. \"reason in at\n# most 4 short steps\" controls both cost and the risk of wandering into\n# unhelpful tangents. Open-ended CoT + tight max_tokens can truncate before\n# the final answer is produced — bound the reasoning or raise the budget.\n\n# [Idiom] CoT + self-consistency (Chapter 11) is a strong combo for high-stakes\n# decisions: generate several independent CoT traces, check if they converge.\n# More robust than trusting a single trace, especially for anything with consequences.\n",[30,1965,1966,1971,1976,1981,1985,1990,1995,2000,2004,2009,2014,2019,2024,2028,2033,2038,2043,2048,2052,2057,2062],{"__ignoreMap":28},[33,1967,1968],{"class":35,"line":36},[33,1969,1970],{"class":39},"# [Idiom] \"Think step by step\" is a floor, not a ceiling. Specify WHAT KIND\n",[33,1972,1973],{"class":35,"line":43},[33,1974,1975],{"class":39},"# of steps: \"list the relevant constraints first, then check each option\n",[33,1977,1978],{"class":35,"line":49},[33,1979,1980],{"class":39},"# against them, then state your conclusion\" gives a specific reasoning scaffold.\n",[33,1982,1983],{"class":35,"line":55},[33,1984,129],{"emptyLinePlaceholder":128},[33,1986,1987],{"class":35,"line":61},[33,1988,1989],{"class":39},"# [Debug] Ask for reasoning BEFORE the answer, never after. Autoregressive\n",[33,1991,1992],{"class":35,"line":67},[33,1993,1994],{"class":39},"# generation means reasoning after the answer is post-hoc justification that\n",[33,1996,1997],{"class":35,"line":73},[33,1998,1999],{"class":39},"# couldn't have influenced the answer.\n",[33,2001,2002],{"class":35,"line":78},[33,2003,129],{"emptyLinePlaceholder":128},[33,2005,2006],{"class":35,"line":84},[33,2007,2008],{"class":39},"# [Debug] Use CoT to debug prompts even if you don't ship it. When a zero-shot\n",[33,2010,2011],{"class":35,"line":90},[33,2012,2013],{"class":39},"# prompt fails mysteriously, add \"think step by step\" and inspect the trace.\n",[33,2015,2016],{"class":35,"line":96},[33,2017,2018],{"class":39},"# It often reveals which assumption or ambiguity is causing the failure (Chapter 4).\n",[33,2020,2021],{"class":35,"line":102},[33,2022,2023],{"class":39},"# Then fix the instruction — sometimes letting you remove the CoT entirely.\n",[33,2025,2026],{"class":35,"line":107},[33,2027,129],{"emptyLinePlaceholder":128},[33,2029,2030],{"class":35,"line":113},[33,2031,2032],{"class":39},"# [Performance] Cap reasoning length for cost-sensitive paths. \"reason in at\n",[33,2034,2035],{"class":35,"line":119},[33,2036,2037],{"class":39},"# most 4 short steps\" controls both cost and the risk of wandering into\n",[33,2039,2040],{"class":35,"line":125},[33,2041,2042],{"class":39},"# unhelpful tangents. Open-ended CoT + tight max_tokens can truncate before\n",[33,2044,2045],{"class":35,"line":132},[33,2046,2047],{"class":39},"# the final answer is produced — bound the reasoning or raise the budget.\n",[33,2049,2050],{"class":35,"line":138},[33,2051,129],{"emptyLinePlaceholder":128},[33,2053,2054],{"class":35,"line":153},[33,2055,2056],{"class":39},"# [Idiom] CoT + self-consistency (Chapter 11) is a strong combo for high-stakes\n",[33,2058,2059],{"class":35,"line":159},[33,2060,2061],{"class":39},"# decisions: generate several independent CoT traces, check if they converge.\n",[33,2063,2064],{"class":35,"line":181},[33,2065,2066],{"class":39},"# More robust than trusting a single trace, especially for anything with consequences.\n",[14,2068,2070],{"id":2069},"️-edge-cases-gotchas","⚠️ Edge Cases & Gotchas",[19,2072,2074],{"filename":2073,"language":22},"edge_cases.py",[24,2075,2077],{"className":26,"code":2076,"language":22,"meta":28,"style":28},"# [Gotcha] Reasoning can be fluent AND wrong simultaneously. A model can produce\n# a chain that reads as completely coherent and confident while containing a\n# subtle error partway through. Because each step conditions the next, one\n# early error propagates and gets \"confirmed\" by everything that follows — the\n# model reasons consistently FROM the mistake rather than toward truth.\n# Fluency of the trace is NOT evidence of correctness. Verify against ground truth.\n\n# [Gotcha] Forcing a reasoning format can truncate the answer at the token limit.\n# If your prompt requests lengthy step-by-step reasoning AND you have a\n# max_tokens limit, the reasoning can consume the whole budget, cutting off\n# before the final answer is ever produced. Either bound the reasoning length,\n# request the answer first (if reasoning is for post-hoc explanation only),\n# or allocate a generous budget with a cheap final-answer format.\n\n# [Gotcha] \"Think step by step\" doesn't guarantee the model uses YOUR intended\n# steps. In zero-shot CoT, the model chooses its own reasoning structure, which\n# may not match yours (it might reason about the wrong sub-problem first).\n# If the SPECIFIC SEQUENCE matters, use few-shot CoT with examples showing\n# that exact sequence.\n\n# [Gotcha] CoT doesn't fix tasks that are about missing information, not missing\n# reasoning. If a prompt asks the model to determine something genuinely\n# un-derivable from the given information, CoT produces confident-looking\n# reasoning that arrives at a number anyway — the scaffold doesn't prevent\n# confabulation of missing premises. See Chapter 17.\n\n# [Gotcha] Extended thinking + prompted CoT can conflict or double up. Enabling\n# a dedicated extended-thinking parameter AND including \"think step by step\"\n# can be redundant (paying for reasoning twice) or interact in undocumented ways.\n# Check current provider docs before combining.\n",[30,2078,2079,2084,2089,2094,2099,2104,2109,2113,2118,2123,2128,2133,2138,2143,2147,2152,2157,2162,2167,2172,2176,2181,2186,2191,2196,2201,2205,2210,2215,2220],{"__ignoreMap":28},[33,2080,2081],{"class":35,"line":36},[33,2082,2083],{"class":39},"# [Gotcha] Reasoning can be fluent AND wrong simultaneously. A model can produce\n",[33,2085,2086],{"class":35,"line":43},[33,2087,2088],{"class":39},"# a chain that reads as completely coherent and confident while containing a\n",[33,2090,2091],{"class":35,"line":49},[33,2092,2093],{"class":39},"# subtle error partway through. Because each step conditions the next, one\n",[33,2095,2096],{"class":35,"line":55},[33,2097,2098],{"class":39},"# early error propagates and gets \"confirmed\" by everything that follows — the\n",[33,2100,2101],{"class":35,"line":61},[33,2102,2103],{"class":39},"# model reasons consistently FROM the mistake rather than toward truth.\n",[33,2105,2106],{"class":35,"line":67},[33,2107,2108],{"class":39},"# Fluency of the trace is NOT evidence of correctness. Verify against ground truth.\n",[33,2110,2111],{"class":35,"line":73},[33,2112,129],{"emptyLinePlaceholder":128},[33,2114,2115],{"class":35,"line":78},[33,2116,2117],{"class":39},"# [Gotcha] Forcing a reasoning format can truncate the answer at the token limit.\n",[33,2119,2120],{"class":35,"line":84},[33,2121,2122],{"class":39},"# If your prompt requests lengthy step-by-step reasoning AND you have a\n",[33,2124,2125],{"class":35,"line":90},[33,2126,2127],{"class":39},"# max_tokens limit, the reasoning can consume the whole budget, cutting off\n",[33,2129,2130],{"class":35,"line":96},[33,2131,2132],{"class":39},"# before the final answer is ever produced. Either bound the reasoning length,\n",[33,2134,2135],{"class":35,"line":102},[33,2136,2137],{"class":39},"# request the answer first (if reasoning is for post-hoc explanation only),\n",[33,2139,2140],{"class":35,"line":107},[33,2141,2142],{"class":39},"# or allocate a generous budget with a cheap final-answer format.\n",[33,2144,2145],{"class":35,"line":113},[33,2146,129],{"emptyLinePlaceholder":128},[33,2148,2149],{"class":35,"line":119},[33,2150,2151],{"class":39},"# [Gotcha] \"Think step by step\" doesn't guarantee the model uses YOUR intended\n",[33,2153,2154],{"class":35,"line":125},[33,2155,2156],{"class":39},"# steps. In zero-shot CoT, the model chooses its own reasoning structure, which\n",[33,2158,2159],{"class":35,"line":132},[33,2160,2161],{"class":39},"# may not match yours (it might reason about the wrong sub-problem first).\n",[33,2163,2164],{"class":35,"line":138},[33,2165,2166],{"class":39},"# If the SPECIFIC SEQUENCE matters, use few-shot CoT with examples showing\n",[33,2168,2169],{"class":35,"line":153},[33,2170,2171],{"class":39},"# that exact sequence.\n",[33,2173,2174],{"class":35,"line":159},[33,2175,129],{"emptyLinePlaceholder":128},[33,2177,2178],{"class":35,"line":181},[33,2179,2180],{"class":39},"# [Gotcha] CoT doesn't fix tasks that are about missing information, not missing\n",[33,2182,2183],{"class":35,"line":193},[33,2184,2185],{"class":39},"# reasoning. If a prompt asks the model to determine something genuinely\n",[33,2187,2188],{"class":35,"line":198},[33,2189,2190],{"class":39},"# un-derivable from the given information, CoT produces confident-looking\n",[33,2192,2193],{"class":35,"line":208},[33,2194,2195],{"class":39},"# reasoning that arrives at a number anyway — the scaffold doesn't prevent\n",[33,2197,2198],{"class":35,"line":214},[33,2199,2200],{"class":39},"# confabulation of missing premises. See Chapter 17.\n",[33,2202,2203],{"class":35,"line":231},[33,2204,129],{"emptyLinePlaceholder":128},[33,2206,2207],{"class":35,"line":251},[33,2208,2209],{"class":39},"# [Gotcha] Extended thinking + prompted CoT can conflict or double up. Enabling\n",[33,2211,2212],{"class":35,"line":261},[33,2213,2214],{"class":39},"# a dedicated extended-thinking parameter AND including \"think step by step\"\n",[33,2216,2217],{"class":35,"line":266},[33,2218,2219],{"class":39},"# can be redundant (paying for reasoning twice) or interact in undocumented ways.\n",[33,2221,2222],{"class":35,"line":272},[33,2223,2224],{"class":39},"# Check current provider docs before combining.\n",[14,2226,2228],{"id":2227},"spot-the-bug","🧠 Spot the Bug",[2230,2231,2232,2233,2235],"p",{},"A developer building a loan pre-approval assistant writes the prompt shown in ",[30,2234,1494],{}," above. The model concludes APPROVED, but the loan-to-income ratio is 5.86x, failing the 5x rule. The model's own reasoning trace claimed to have checked this and stated it passed. What does this reveal?",[2237,2238,2239,2243,2251],"details",{},[2240,2241,2242],"summary",{},"Answer",[2230,2244,2245,2246,2250],{},"The reasoning trace ",[2247,2248,2249],"em",{},"looked"," like it verified the constraint, but the underlying multiplication either wasn't performed correctly or was misreported. This is the \"fluent and wrong\" failure: a chain that reads as if it checked something can still get that specific check wrong, especially for multi-digit arithmetic (Chapter 1's tokenization problem).",[2230,2252,2253,2254,2258,2259,2262,2263,2265],{},"The deeper lesson: ",[2255,2256,2257],"strong",{},"a visible reasoning trace is not a verification mechanism",". For high-stakes decisions like loan approval, the fix is not \"add more CoT instructions\" but to ",[2255,2260,2261],{},"externalize the arithmetic and threshold checks into deterministic code"," (see ",[30,2264,1545],{},") and use the model only for parts that genuinely require judgment — with CoT as an explanation layer on top of verified numbers, not as the source of truth for the numbers themselves.",[14,2267,2269],{"id":2268},"key-takeaways","Key Takeaways",[19,2271,2273],{"filename":2272,"language":22},"key_takeaways.py",[24,2274,2276],{"className":26,"code":2275,"language":22,"meta":28,"style":28},"\"\"\"\nChain-of-thought — mechanism, application, and limits.\n\"\"\"\n\n# 1. CoT works because autoregressive generation lets earlier reasoning tokens\n#    condition and improve later ones. It's self-generated context, not \"thinking\n#    harder.\" The model uses its own output as working memory.\n#    without_cot: answer = model(prompt + \"Answer:\")\n#    with_cot:    answer = model(prompt + reasoning + \"Answer:\")  # reasoning is context\n\n# 2. Zero-shot CoT (\"think step by step\") is a cheap default for multi-step tasks.\n#    Few-shot CoT (demonstrating the reasoning pattern) gives control over the\n#    specific STRUCTURE of the reasoning.\n\n# 3. CoT helps most on derivation-based tasks (arithmetic, logic, planning).\n#    It hurts on simple lookups, classifications, and creative tasks — adds\n#    latency and cost with no accuracy gain, can cause overthinking.\n\n# 4. A fluent, confident reasoning trace is NOT proof of correctness. Reasoning\n#    can be internally consistent and still wrong, especially on arithmetic.\n#    → High-stakes numeric decisions: use deterministic code, not CoT, for the numbers.\n\n# 5. Extended thinking ≠ prompted CoT. Dedicated reasoning modes (API-level\n#    settings) are trained and optimized for deep reasoning. Use the dedicated\n#    mechanism when available; reserve prompted CoT for when you need a SPECIFIC\n#    reasoning structure the model wouldn't produce on its own.\n",[30,2277,2278,2282,2287,2291,2295,2300,2305,2310,2315,2320,2324,2329,2334,2339,2343,2348,2353,2358,2362,2367,2372,2377,2381,2386,2391,2396],{"__ignoreMap":28},[33,2279,2280],{"class":35,"line":36},[33,2281,879],{"class":174},[33,2283,2284],{"class":35,"line":43},[33,2285,2286],{"class":174},"Chain-of-thought — mechanism, application, and limits.\n",[33,2288,2289],{"class":35,"line":49},[33,2290,879],{"class":174},[33,2292,2293],{"class":35,"line":55},[33,2294,129],{"emptyLinePlaceholder":128},[33,2296,2297],{"class":35,"line":61},[33,2298,2299],{"class":39},"# 1. CoT works because autoregressive generation lets earlier reasoning tokens\n",[33,2301,2302],{"class":35,"line":67},[33,2303,2304],{"class":39},"#    condition and improve later ones. It's self-generated context, not \"thinking\n",[33,2306,2307],{"class":35,"line":73},[33,2308,2309],{"class":39},"#    harder.\" The model uses its own output as working memory.\n",[33,2311,2312],{"class":35,"line":78},[33,2313,2314],{"class":39},"#    without_cot: answer = model(prompt + \"Answer:\")\n",[33,2316,2317],{"class":35,"line":84},[33,2318,2319],{"class":39},"#    with_cot:    answer = model(prompt + reasoning + \"Answer:\")  # reasoning is context\n",[33,2321,2322],{"class":35,"line":90},[33,2323,129],{"emptyLinePlaceholder":128},[33,2325,2326],{"class":35,"line":96},[33,2327,2328],{"class":39},"# 2. Zero-shot CoT (\"think step by step\") is a cheap default for multi-step tasks.\n",[33,2330,2331],{"class":35,"line":102},[33,2332,2333],{"class":39},"#    Few-shot CoT (demonstrating the reasoning pattern) gives control over the\n",[33,2335,2336],{"class":35,"line":107},[33,2337,2338],{"class":39},"#    specific STRUCTURE of the reasoning.\n",[33,2340,2341],{"class":35,"line":113},[33,2342,129],{"emptyLinePlaceholder":128},[33,2344,2345],{"class":35,"line":119},[33,2346,2347],{"class":39},"# 3. CoT helps most on derivation-based tasks (arithmetic, logic, planning).\n",[33,2349,2350],{"class":35,"line":125},[33,2351,2352],{"class":39},"#    It hurts on simple lookups, classifications, and creative tasks — adds\n",[33,2354,2355],{"class":35,"line":132},[33,2356,2357],{"class":39},"#    latency and cost with no accuracy gain, can cause overthinking.\n",[33,2359,2360],{"class":35,"line":138},[33,2361,129],{"emptyLinePlaceholder":128},[33,2363,2364],{"class":35,"line":153},[33,2365,2366],{"class":39},"# 4. A fluent, confident reasoning trace is NOT proof of correctness. Reasoning\n",[33,2368,2369],{"class":35,"line":159},[33,2370,2371],{"class":39},"#    can be internally consistent and still wrong, especially on arithmetic.\n",[33,2373,2374],{"class":35,"line":181},[33,2375,2376],{"class":39},"#    → High-stakes numeric decisions: use deterministic code, not CoT, for the numbers.\n",[33,2378,2379],{"class":35,"line":193},[33,2380,129],{"emptyLinePlaceholder":128},[33,2382,2383],{"class":35,"line":198},[33,2384,2385],{"class":39},"# 5. Extended thinking ≠ prompted CoT. Dedicated reasoning modes (API-level\n",[33,2387,2388],{"class":35,"line":208},[33,2389,2390],{"class":39},"#    settings) are trained and optimized for deep reasoning. Use the dedicated\n",[33,2392,2393],{"class":35,"line":214},[33,2394,2395],{"class":39},"#    mechanism when available; reserve prompted CoT for when you need a SPECIFIC\n",[33,2397,2398],{"class":35,"line":231},[33,2399,2400],{"class":39},"#    reasoning structure the model wouldn't produce on its own.\n",[2402,2403,2404],"style",{},"html pre.shiki code .sdCPZ, html code.shiki .sdCPZ{--shiki-default:#6A737D;--shiki-github-dark:#6A737D}html pre.shiki code .svdQ7, html code.shiki .svdQ7{--shiki-default:#D73A49;--shiki-github-dark:#F97583}html pre.shiki code .sIsaT, html code.shiki .sIsaT{--shiki-default:#6F42C1;--shiki-github-dark:#B392F0}html pre.shiki code .ssxIu, html code.shiki .ssxIu{--shiki-default:#24292E;--shiki-github-dark:#E1E4E8}html pre.shiki code .sJ6F3, html code.shiki .sJ6F3{--shiki-default:#032F62;--shiki-github-dark:#9ECBFF}html .default .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .github-dark .shiki span {color: var(--shiki-github-dark);background: var(--shiki-github-dark-bg);font-style: var(--shiki-github-dark-font-style);font-weight: var(--shiki-github-dark-font-weight);text-decoration: var(--shiki-github-dark-text-decoration);}html.github-dark .shiki span {color: var(--shiki-github-dark);background: var(--shiki-github-dark-bg);font-style: var(--shiki-github-dark-font-style);font-weight: var(--shiki-github-dark-font-weight);text-decoration: var(--shiki-github-dark-text-decoration);}html pre.shiki code .sCrzJ, html code.shiki .sCrzJ{--shiki-default:#E36209;--shiki-github-dark:#FFAB70}html pre.shiki code .snvgF, html code.shiki .snvgF{--shiki-default:#005CC5;--shiki-github-dark:#79B8FF}",{"title":28,"searchDepth":43,"depth":43,"links":2406},[2407,2408,2409,2410,2411,2412,2413,2414,2415,2416,2417],{"id":16,"depth":43,"text":17},{"id":290,"depth":43,"text":291},{"id":370,"depth":43,"text":371},{"id":459,"depth":43,"text":460},{"id":818,"depth":43,"text":819},{"id":1010,"depth":43,"text":1011},{"id":1490,"depth":43,"text":1491},{"id":1956,"depth":43,"text":1957},{"id":2069,"depth":43,"text":2070},{"id":2227,"depth":43,"text":2228},{"id":2268,"depth":43,"text":2269},"CoT as self-generated context narrowing — zero-shot, few-shot, extended thinking, reasoning-answer separation, and the fluent-but-wrong failure mode. Production patterns for bounded reasoning and high-stakes verification. Code-first reference for mid-to-senior engineers.","md",{},"\u002Fprompt-engineering\u002F05-chain-of-thought-prompting",{"title":5,"description":2418},"prompt-engineering\u002F05-chain-of-thought-prompting","6602oEr_4YFmVJIIAGUIDbWZWlBcQ_LmXpQQ6FQhnG8",1789924650824]