[{"data":1,"prerenderedAt":5151},["ShallowReactive",2],{"page-\u002Fprompt-engineering\u002F01-introduction-and-how-llms-work":3},{"id":4,"title":5,"body":6,"description":5144,"extension":5145,"meta":5146,"navigation":77,"path":5147,"seo":5148,"stem":5149,"__hash__":5150},"content\u002Fprompt-engineering\u002F01-introduction-and-how-llms-work.md","01 — Introduction & How LLMs Work",{"type":7,"value":8,"toc":5132},"minimark",[9,13,18,890,902,906,1319,1323,1951,1955,2958,2962,3266,3270,3372,4080,4084,4285,4289,4745,4749,4752,4787,4800,4973,4977,5128],[10,11,5],"h1",{"id":12},"_01-introduction-how-llms-work",[14,15,17],"h2",{"id":16},"the-core-mechanism-next-token-prediction","The Core Mechanism: Next-Token Prediction",[19,20,23],"code-wrapper",{"filename":21,"language":22},"next_token_prediction.py","python",[24,25,29],"pre",{"className":26,"code":27,"language":22,"meta":28,"style":28},"language-python shiki shiki-themes github-light github-dark","import torch\nfrom torch.nn.functional import softmax, log_softmax\nfrom dataclasses import dataclass\n\n@dataclass\nclass GenerationConfig:\n    \"\"\"Sampling parameters that shape the probability distribution at each step.\"\"\"\n    temperature: float = 1.0       # >1 flattens (more random), \u003C1 sharpens (more deterministic)\n    top_k: int = 0                 # 0 = disabled; otherwise only consider top-k logits\n    top_p: float = 1.0             # 1.0 = disabled; otherwise nucleus sampling — smallest set whose cumulative prob >= top_p\n    max_tokens: int = 512          # hard stop regardless of EOS\n    stop_sequences: tuple[str, ...] = ()  # early termination on exact string match\n\ndef generate_next_token(\n    model,                        # frozen transformer — weights never change during inference\n    token_ids: torch.Tensor,      # shape: [seq_len] — the full context so far, including model's own output\n    config: GenerationConfig,\n    is_first_token: bool = False,  # first generation step after the user prompt\n) -> int:\n    \"\"\"\n    The ONE thing an LLM does: given a token sequence, produce the next token id.\n    Everything — reasoning, coding, refusal, translation — is this function called in a loop.\n    \"\"\"\n    with torch.no_grad():                # inference only — no gradient computation, no weight updates\n        logits = model(token_ids)        # [vocab_size] — raw unnormalized scores for every possible next token\n\n    # --- Temperature: scales logits before softmax ---\n    # temperature=0.5 makes high-probability tokens even more likely (sharper distribution)\n    # temperature=2.0 flattens, making low-probability tokens more likely (more creative \u002F noisier)\n    if config.temperature != 1.0:\n        logits = logits \u002F config.temperature   # divide logits by T; T\u003C1 amplifies differences, T>1 dampens them\n\n    # --- Top-k: keep only the k highest-scoring tokens, set rest to -inf ---\n    if config.top_k > 0:\n        top_vals, _ = torch.topk(logits, config.top_k)\n        min_val = top_vals[-1]                       # the k-th highest logit\n        logits = torch.where(logits \u003C min_val, torch.tensor(float('-inf')), logits)\n\n    # --- Top-p (nucleus): keep smallest set of tokens whose cumulative prob >= top_p ---\n    if config.top_p \u003C 1.0:\n        sorted_logits, sorted_idx = torch.sort(logits, descending=True)\n        cumulative_probs = torch.cumsum(softmax(sorted_logits, dim=-1), dim=-1)\n        # mask tokens that fall outside the nucleus (past the cumulative threshold)\n        sorted_mask = cumulative_probs - softmax(sorted_logits, dim=-1) > config.top_p\n        sorted_logits[sorted_mask] = float('-inf')\n        logits = torch.full_like(logits, float('-inf'))\n        logits[sorted_idx] = sorted_logits          # scatter back to original positions\n\n    # --- Sample from the (possibly filtered, temperature-scaled) distribution ---\n    probs = softmax(logits, dim=-1)                 # normalize to a valid probability distribution\n    next_token_id = torch.multinomial(probs, num_samples=1).item()\n\n    # NOTE: even with temperature=0 (greedy = argmax), floating-point non-determinism\n    # across GPU batches means you CANNOT guarantee bit-identical output run-to-run.\n    # Never build a system that assumes exact reproducibility from an LLM.\n    return next_token_id\n\ndef generate_loop(model, prompt_ids: list[int], config: GenerationConfig) -> list[int]:\n    \"\"\"The autoregressive loop: append, re-feed, repeat. Each token depends on ALL prior tokens.\"\"\"\n    token_ids = torch.tensor(prompt_ids)\n    generated: list[int] = []\n    for i in range(config.max_tokens):\n        next_id = generate_next_token(model, token_ids, config, is_first_token=(i == 0))\n        if next_id == model.eos_token_id:            # end-of-turn marker — model decided it's done\n            break\n        generated.append(next_id)\n        token_ids = torch.cat([token_ids, torch.tensor([next_id])])\n        # CRITICAL: the model conditions on its OWN output as it generates.\n        # If it starts hedging (\"I'm not sure, but...\"), those tokens make further hedging\n        # more probable — the model can \"talk itself into\" a stance it wouldn't have started with.\n    return generated\n","",[30,31,32,45,59,72,79,86,98,105,125,142,157,173,200,205,217,226,235,241,261,271,277,283,289,294,306,320,325,331,337,343,359,378,383,389,404,415,438,465,470,476,490,513,543,549,579,596,615,629,634,640,663,684,689,701,707,713,722,727,748,754,765,780,798,824,841,847,853,864,870,876,882],"code",{"__ignoreMap":28},[33,34,37,41],"span",{"class":35,"line":36},"line",1,[33,38,40],{"class":39},"svdQ7","import",[33,42,44],{"class":43},"ssxIu"," torch\n",[33,46,48,51,54,56],{"class":35,"line":47},2,[33,49,50],{"class":39},"from",[33,52,53],{"class":43}," torch.nn.functional ",[33,55,40],{"class":39},[33,57,58],{"class":43}," softmax, log_softmax\n",[33,60,62,64,67,69],{"class":35,"line":61},3,[33,63,50],{"class":39},[33,65,66],{"class":43}," dataclasses ",[33,68,40],{"class":39},[33,70,71],{"class":43}," dataclass\n",[33,73,75],{"class":35,"line":74},4,[33,76,78],{"emptyLinePlaceholder":77},true,"\n",[33,80,82],{"class":35,"line":81},5,[33,83,85],{"class":84},"sIsaT","@dataclass\n",[33,87,89,92,95],{"class":35,"line":88},6,[33,90,91],{"class":39},"class",[33,93,94],{"class":84}," GenerationConfig",[33,96,97],{"class":43},":\n",[33,99,101],{"class":35,"line":100},7,[33,102,104],{"class":103},"sJ6F3","    \"\"\"Sampling parameters that shape the probability distribution at each step.\"\"\"\n",[33,106,108,111,115,118,121],{"class":35,"line":107},8,[33,109,110],{"class":43},"    temperature: ",[33,112,114],{"class":113},"snvgF","float",[33,116,117],{"class":39}," =",[33,119,120],{"class":113}," 1.0",[33,122,124],{"class":123},"sdCPZ","       # >1 flattens (more random), \u003C1 sharpens (more deterministic)\n",[33,126,128,131,134,136,139],{"class":35,"line":127},9,[33,129,130],{"class":43},"    top_k: ",[33,132,133],{"class":113},"int",[33,135,117],{"class":39},[33,137,138],{"class":113}," 0",[33,140,141],{"class":123},"                 # 0 = disabled; otherwise only consider top-k logits\n",[33,143,145,148,150,152,154],{"class":35,"line":144},10,[33,146,147],{"class":43},"    top_p: ",[33,149,114],{"class":113},[33,151,117],{"class":39},[33,153,120],{"class":113},[33,155,156],{"class":123},"             # 1.0 = disabled; otherwise nucleus sampling — smallest set whose cumulative prob >= top_p\n",[33,158,160,163,165,167,170],{"class":35,"line":159},11,[33,161,162],{"class":43},"    max_tokens: ",[33,164,133],{"class":113},[33,166,117],{"class":39},[33,168,169],{"class":113}," 512",[33,171,172],{"class":123},"          # hard stop regardless of EOS\n",[33,174,176,179,182,185,188,191,194,197],{"class":35,"line":175},12,[33,177,178],{"class":43},"    stop_sequences: tuple[",[33,180,181],{"class":113},"str",[33,183,184],{"class":43},", ",[33,186,187],{"class":113},"...",[33,189,190],{"class":43},"] ",[33,192,193],{"class":39},"=",[33,195,196],{"class":43}," ()  ",[33,198,199],{"class":123},"# early termination on exact string match\n",[33,201,203],{"class":35,"line":202},13,[33,204,78],{"emptyLinePlaceholder":77},[33,206,208,211,214],{"class":35,"line":207},14,[33,209,210],{"class":39},"def",[33,212,213],{"class":84}," generate_next_token",[33,215,216],{"class":43},"(\n",[33,218,220,223],{"class":35,"line":219},15,[33,221,222],{"class":43},"    model,                        ",[33,224,225],{"class":123},"# frozen transformer — weights never change during inference\n",[33,227,229,232],{"class":35,"line":228},16,[33,230,231],{"class":43},"    token_ids: torch.Tensor,      ",[33,233,234],{"class":123},"# shape: [seq_len] — the full context so far, including model's own output\n",[33,236,238],{"class":35,"line":237},17,[33,239,240],{"class":43},"    config: GenerationConfig,\n",[33,242,244,247,250,252,255,258],{"class":35,"line":243},18,[33,245,246],{"class":43},"    is_first_token: ",[33,248,249],{"class":113},"bool",[33,251,117],{"class":39},[33,253,254],{"class":113}," False",[33,256,257],{"class":43},",  ",[33,259,260],{"class":123},"# first generation step after the user prompt\n",[33,262,264,267,269],{"class":35,"line":263},19,[33,265,266],{"class":43},") -> ",[33,268,133],{"class":113},[33,270,97],{"class":43},[33,272,274],{"class":35,"line":273},20,[33,275,276],{"class":103},"    \"\"\"\n",[33,278,280],{"class":35,"line":279},21,[33,281,282],{"class":103},"    The ONE thing an LLM does: given a token sequence, produce the next token id.\n",[33,284,286],{"class":35,"line":285},22,[33,287,288],{"class":103},"    Everything — reasoning, coding, refusal, translation — is this function called in a loop.\n",[33,290,292],{"class":35,"line":291},23,[33,293,276],{"class":103},[33,295,297,300,303],{"class":35,"line":296},24,[33,298,299],{"class":39},"    with",[33,301,302],{"class":43}," torch.no_grad():                ",[33,304,305],{"class":123},"# inference only — no gradient computation, no weight updates\n",[33,307,309,312,314,317],{"class":35,"line":308},25,[33,310,311],{"class":43},"        logits ",[33,313,193],{"class":39},[33,315,316],{"class":43}," model(token_ids)        ",[33,318,319],{"class":123},"# [vocab_size] — raw unnormalized scores for every possible next token\n",[33,321,323],{"class":35,"line":322},26,[33,324,78],{"emptyLinePlaceholder":77},[33,326,328],{"class":35,"line":327},27,[33,329,330],{"class":123},"    # --- Temperature: scales logits before softmax ---\n",[33,332,334],{"class":35,"line":333},28,[33,335,336],{"class":123},"    # temperature=0.5 makes high-probability tokens even more likely (sharper distribution)\n",[33,338,340],{"class":35,"line":339},29,[33,341,342],{"class":123},"    # temperature=2.0 flattens, making low-probability tokens more likely (more creative \u002F noisier)\n",[33,344,346,349,352,355,357],{"class":35,"line":345},30,[33,347,348],{"class":39},"    if",[33,350,351],{"class":43}," config.temperature ",[33,353,354],{"class":39},"!=",[33,356,120],{"class":113},[33,358,97],{"class":43},[33,360,362,364,366,369,372,375],{"class":35,"line":361},31,[33,363,311],{"class":43},[33,365,193],{"class":39},[33,367,368],{"class":43}," logits ",[33,370,371],{"class":39},"\u002F",[33,373,374],{"class":43}," config.temperature   ",[33,376,377],{"class":123},"# divide logits by T; T\u003C1 amplifies differences, T>1 dampens them\n",[33,379,381],{"class":35,"line":380},32,[33,382,78],{"emptyLinePlaceholder":77},[33,384,386],{"class":35,"line":385},33,[33,387,388],{"class":123},"    # --- Top-k: keep only the k highest-scoring tokens, set rest to -inf ---\n",[33,390,392,394,397,400,402],{"class":35,"line":391},34,[33,393,348],{"class":39},[33,395,396],{"class":43}," config.top_k ",[33,398,399],{"class":39},">",[33,401,138],{"class":113},[33,403,97],{"class":43},[33,405,407,410,412],{"class":35,"line":406},35,[33,408,409],{"class":43},"        top_vals, _ ",[33,411,193],{"class":39},[33,413,414],{"class":43}," torch.topk(logits, config.top_k)\n",[33,416,418,421,423,426,429,432,435],{"class":35,"line":417},36,[33,419,420],{"class":43},"        min_val ",[33,422,193],{"class":39},[33,424,425],{"class":43}," top_vals[",[33,427,428],{"class":39},"-",[33,430,431],{"class":113},"1",[33,433,434],{"class":43},"]                       ",[33,436,437],{"class":123},"# the k-th highest logit\n",[33,439,441,443,445,448,451,454,456,459,462],{"class":35,"line":440},37,[33,442,311],{"class":43},[33,444,193],{"class":39},[33,446,447],{"class":43}," torch.where(logits ",[33,449,450],{"class":39},"\u003C",[33,452,453],{"class":43}," min_val, torch.tensor(",[33,455,114],{"class":113},[33,457,458],{"class":43},"(",[33,460,461],{"class":103},"'-inf'",[33,463,464],{"class":43},")), logits)\n",[33,466,468],{"class":35,"line":467},38,[33,469,78],{"emptyLinePlaceholder":77},[33,471,473],{"class":35,"line":472},39,[33,474,475],{"class":123},"    # --- Top-p (nucleus): keep smallest set of tokens whose cumulative prob >= top_p ---\n",[33,477,479,481,484,486,488],{"class":35,"line":478},40,[33,480,348],{"class":39},[33,482,483],{"class":43}," config.top_p ",[33,485,450],{"class":39},[33,487,120],{"class":113},[33,489,97],{"class":43},[33,491,493,496,498,501,505,507,510],{"class":35,"line":492},41,[33,494,495],{"class":43},"        sorted_logits, sorted_idx ",[33,497,193],{"class":39},[33,499,500],{"class":43}," torch.sort(logits, ",[33,502,504],{"class":503},"sCrzJ","descending",[33,506,193],{"class":39},[33,508,509],{"class":113},"True",[33,511,512],{"class":43},")\n",[33,514,516,519,521,524,527,530,532,535,537,539,541],{"class":35,"line":515},42,[33,517,518],{"class":43},"        cumulative_probs ",[33,520,193],{"class":39},[33,522,523],{"class":43}," torch.cumsum(softmax(sorted_logits, ",[33,525,526],{"class":503},"dim",[33,528,529],{"class":39},"=-",[33,531,431],{"class":113},[33,533,534],{"class":43},"), ",[33,536,526],{"class":503},[33,538,529],{"class":39},[33,540,431],{"class":113},[33,542,512],{"class":43},[33,544,546],{"class":35,"line":545},43,[33,547,548],{"class":123},"        # mask tokens that fall outside the nucleus (past the cumulative threshold)\n",[33,550,552,555,557,560,562,565,567,569,571,574,576],{"class":35,"line":551},44,[33,553,554],{"class":43},"        sorted_mask ",[33,556,193],{"class":39},[33,558,559],{"class":43}," cumulative_probs ",[33,561,428],{"class":39},[33,563,564],{"class":43}," softmax(sorted_logits, ",[33,566,526],{"class":503},[33,568,529],{"class":39},[33,570,431],{"class":113},[33,572,573],{"class":43},") ",[33,575,399],{"class":39},[33,577,578],{"class":43}," config.top_p\n",[33,580,582,585,587,590,592,594],{"class":35,"line":581},45,[33,583,584],{"class":43},"        sorted_logits[sorted_mask] ",[33,586,193],{"class":39},[33,588,589],{"class":113}," float",[33,591,458],{"class":43},[33,593,461],{"class":103},[33,595,512],{"class":43},[33,597,599,601,603,606,608,610,612],{"class":35,"line":598},46,[33,600,311],{"class":43},[33,602,193],{"class":39},[33,604,605],{"class":43}," torch.full_like(logits, ",[33,607,114],{"class":113},[33,609,458],{"class":43},[33,611,461],{"class":103},[33,613,614],{"class":43},"))\n",[33,616,618,621,623,626],{"class":35,"line":617},47,[33,619,620],{"class":43},"        logits[sorted_idx] ",[33,622,193],{"class":39},[33,624,625],{"class":43}," sorted_logits          ",[33,627,628],{"class":123},"# scatter back to original positions\n",[33,630,632],{"class":35,"line":631},48,[33,633,78],{"emptyLinePlaceholder":77},[33,635,637],{"class":35,"line":636},49,[33,638,639],{"class":123},"    # --- Sample from the (possibly filtered, temperature-scaled) distribution ---\n",[33,641,643,646,648,651,653,655,657,660],{"class":35,"line":642},50,[33,644,645],{"class":43},"    probs ",[33,647,193],{"class":39},[33,649,650],{"class":43}," softmax(logits, ",[33,652,526],{"class":503},[33,654,529],{"class":39},[33,656,431],{"class":113},[33,658,659],{"class":43},")                 ",[33,661,662],{"class":123},"# normalize to a valid probability distribution\n",[33,664,666,669,671,674,677,679,681],{"class":35,"line":665},51,[33,667,668],{"class":43},"    next_token_id ",[33,670,193],{"class":39},[33,672,673],{"class":43}," torch.multinomial(probs, ",[33,675,676],{"class":503},"num_samples",[33,678,193],{"class":39},[33,680,431],{"class":113},[33,682,683],{"class":43},").item()\n",[33,685,687],{"class":35,"line":686},52,[33,688,78],{"emptyLinePlaceholder":77},[33,690,692,695,698],{"class":35,"line":691},53,[33,693,694],{"class":123},"    # ",[33,696,697],{"class":39},"NOTE",[33,699,700],{"class":123},": even with temperature=0 (greedy = argmax), floating-point non-determinism\n",[33,702,704],{"class":35,"line":703},54,[33,705,706],{"class":123},"    # across GPU batches means you CANNOT guarantee bit-identical output run-to-run.\n",[33,708,710],{"class":35,"line":709},55,[33,711,712],{"class":123},"    # Never build a system that assumes exact reproducibility from an LLM.\n",[33,714,716,719],{"class":35,"line":715},56,[33,717,718],{"class":39},"    return",[33,720,721],{"class":43}," next_token_id\n",[33,723,725],{"class":35,"line":724},57,[33,726,78],{"emptyLinePlaceholder":77},[33,728,730,732,735,738,740,743,745],{"class":35,"line":729},58,[33,731,210],{"class":39},[33,733,734],{"class":84}," generate_loop",[33,736,737],{"class":43},"(model, prompt_ids: list[",[33,739,133],{"class":113},[33,741,742],{"class":43},"], config: GenerationConfig) -> list[",[33,744,133],{"class":113},[33,746,747],{"class":43},"]:\n",[33,749,751],{"class":35,"line":750},59,[33,752,753],{"class":103},"    \"\"\"The autoregressive loop: append, re-feed, repeat. Each token depends on ALL prior tokens.\"\"\"\n",[33,755,757,760,762],{"class":35,"line":756},60,[33,758,759],{"class":43},"    token_ids ",[33,761,193],{"class":39},[33,763,764],{"class":43}," torch.tensor(prompt_ids)\n",[33,766,768,771,773,775,777],{"class":35,"line":767},61,[33,769,770],{"class":43},"    generated: list[",[33,772,133],{"class":113},[33,774,190],{"class":43},[33,776,193],{"class":39},[33,778,779],{"class":43}," []\n",[33,781,783,786,789,792,795],{"class":35,"line":782},62,[33,784,785],{"class":39},"    for",[33,787,788],{"class":43}," i ",[33,790,791],{"class":39},"in",[33,793,794],{"class":113}," range",[33,796,797],{"class":43},"(config.max_tokens):\n",[33,799,801,804,806,809,812,814,817,820,822],{"class":35,"line":800},63,[33,802,803],{"class":43},"        next_id ",[33,805,193],{"class":39},[33,807,808],{"class":43}," generate_next_token(model, token_ids, config, ",[33,810,811],{"class":503},"is_first_token",[33,813,193],{"class":39},[33,815,816],{"class":43},"(i ",[33,818,819],{"class":39},"==",[33,821,138],{"class":113},[33,823,614],{"class":43},[33,825,827,830,833,835,838],{"class":35,"line":826},64,[33,828,829],{"class":39},"        if",[33,831,832],{"class":43}," next_id ",[33,834,819],{"class":39},[33,836,837],{"class":43}," model.eos_token_id:            ",[33,839,840],{"class":123},"# end-of-turn marker — model decided it's done\n",[33,842,844],{"class":35,"line":843},65,[33,845,846],{"class":39},"            break\n",[33,848,850],{"class":35,"line":849},66,[33,851,852],{"class":43},"        generated.append(next_id)\n",[33,854,856,859,861],{"class":35,"line":855},67,[33,857,858],{"class":43},"        token_ids ",[33,860,193],{"class":39},[33,862,863],{"class":43}," torch.cat([token_ids, torch.tensor([next_id])])\n",[33,865,867],{"class":35,"line":866},68,[33,868,869],{"class":123},"        # CRITICAL: the model conditions on its OWN output as it generates.\n",[33,871,873],{"class":35,"line":872},69,[33,874,875],{"class":123},"        # If it starts hedging (\"I'm not sure, but...\"), those tokens make further hedging\n",[33,877,879],{"class":35,"line":878},70,[33,880,881],{"class":123},"        # more probable — the model can \"talk itself into\" a stance it wouldn't have started with.\n",[33,883,885,887],{"class":35,"line":884},71,[33,886,718],{"class":39},[33,888,889],{"class":43}," generated\n",[891,892,893,894,897,898,901],"p",{},"The entire field of prompt engineering targets the ",[30,895,896],{},"prompt_ids"," input to this loop. Every technique in this course — few-shot examples, chain-of-thought, system prompts, XML structuring — is a way of shaping the token sequence that enters ",[30,899,900],{},"generate_loop",", which in turn shapes the probability distribution sampled at each step.",[14,903,905],{"id":904},"training-vs-inference","Training vs. Inference",[19,907,909],{"filename":908,"language":22},"training_vs_inference.py",[24,910,912],{"className":26,"code":911,"language":22,"meta":28,"style":28},"from dataclasses import dataclass\n\n@dataclass\nclass TrainingPhase:\n    \"\"\"What happens BEFORE you ever send a request. You control NOTHING here.\"\"\"\n    objective: str = \"next-token prediction via gradient descent\"\n    # The model's weights are adjusted over billions of text examples so that\n    # P(next_token | context) increasingly matches the training distribution.\n    alignment: tuple[str, ...] = (\"RLHF\", \"DPO\", \"constitutional_ai\")\n    # After base pretraining, fine-tuning + human-feedback alignment shapes\n    # instruction-following, refusal behavior, and helpfulness.\n    weights_change: bool = True              # ← THIS is the defining difference\n    you_control: tuple[str, ...] = ()        # nothing — the provider already did this\n\n@dataclass\nclass InferencePhase:\n    \"\"\"What happens when YOU send a request. You control EVERYTHING here.\"\"\"\n    weights_change: bool = False             # ← frozen. The model learns NOTHING new.\n    # \"In-context learning\" is a misnomer — no learning occurs. You're selecting\n    # which pre-trained distribution region to sample from via the tokens you supply.\n    you_control: tuple[str, ...] = (\n        \"system_prompt\",       # conditions the distribution toward a role\u002Fbehavior region\n        \"user_prompt\",         # the actual task\n        \"few_shot_examples\",   # demonstrates the input→output mapping in-context\n        \"temperature\",         # sampling entropy\n        \"top_k\", \"top_p\",      # truncation of the candidate token set\n        \"max_tokens\",          # generation budget\n        \"stop_sequences\",      # early termination triggers\n    )\n\n# --- The key distinction for prompt engineers ---\n# Training: model.weights \u003C- gradient_step(loss(predictions, targets))\n#   You weren't there. The model's capability ceiling is already fixed.\n# Inference: output = sample(model.forward(your_tokens))\n#   model.weights are READ-ONLY. Your ONLY lever is the token sequence you send.\n#   A well-crafted few-shot prompt can approximate fine-tuned behavior —\n#   without touching a single weight. This is why prompting is powerful.\n\n@dataclass\nclass ChatMemoryIllusion:\n    \"\"\"Explains why 'chat memory' is not real memory.\"\"\"\n    # Every API call is stateless. The model has NO persistent state between calls.\n    # What products call \"memory\" is actually:\n    #   1. Your client code stores prior messages\n    #   2. On each new request, the ENTIRE transcript is re-sent as input tokens\n    #   3. The model re-reads everything from scratch — it has no idea it \"remembers\"\n    # This means: longer \"memory\" = more input tokens = more cost + latency per turn\n    # AND eventually the transcript exceeds the context window and old messages\n    # must be dropped, summarized, or truncated — silently, if you're not careful.\n    api_is_stateless: bool = True\n    memory_mechanism: str = \"client resends full transcript as tokens each request\"\n",[30,913,914,924,928,932,941,946,958,963,968,1001,1006,1011,1026,1047,1051,1055,1064,1069,1082,1087,1092,1109,1120,1131,1142,1152,1168,1179,1189,1194,1198,1203,1208,1213,1218,1223,1228,1233,1237,1241,1250,1255,1260,1265,1270,1275,1280,1285,1290,1295,1307],{"__ignoreMap":28},[33,915,916,918,920,922],{"class":35,"line":36},[33,917,50],{"class":39},[33,919,66],{"class":43},[33,921,40],{"class":39},[33,923,71],{"class":43},[33,925,926],{"class":35,"line":47},[33,927,78],{"emptyLinePlaceholder":77},[33,929,930],{"class":35,"line":61},[33,931,85],{"class":84},[33,933,934,936,939],{"class":35,"line":74},[33,935,91],{"class":39},[33,937,938],{"class":84}," TrainingPhase",[33,940,97],{"class":43},[33,942,943],{"class":35,"line":81},[33,944,945],{"class":103},"    \"\"\"What happens BEFORE you ever send a request. You control NOTHING here.\"\"\"\n",[33,947,948,951,953,955],{"class":35,"line":88},[33,949,950],{"class":43},"    objective: ",[33,952,181],{"class":113},[33,954,117],{"class":39},[33,956,957],{"class":103}," \"next-token prediction via gradient descent\"\n",[33,959,960],{"class":35,"line":100},[33,961,962],{"class":123},"    # The model's weights are adjusted over billions of text examples so that\n",[33,964,965],{"class":35,"line":107},[33,966,967],{"class":123},"    # P(next_token | context) increasingly matches the training distribution.\n",[33,969,970,973,975,977,979,981,983,986,989,991,994,996,999],{"class":35,"line":127},[33,971,972],{"class":43},"    alignment: tuple[",[33,974,181],{"class":113},[33,976,184],{"class":43},[33,978,187],{"class":113},[33,980,190],{"class":43},[33,982,193],{"class":39},[33,984,985],{"class":43}," (",[33,987,988],{"class":103},"\"RLHF\"",[33,990,184],{"class":43},[33,992,993],{"class":103},"\"DPO\"",[33,995,184],{"class":43},[33,997,998],{"class":103},"\"constitutional_ai\"",[33,1000,512],{"class":43},[33,1002,1003],{"class":35,"line":144},[33,1004,1005],{"class":123},"    # After base pretraining, fine-tuning + human-feedback alignment shapes\n",[33,1007,1008],{"class":35,"line":159},[33,1009,1010],{"class":123},"    # instruction-following, refusal behavior, and helpfulness.\n",[33,1012,1013,1016,1018,1020,1023],{"class":35,"line":175},[33,1014,1015],{"class":43},"    weights_change: ",[33,1017,249],{"class":113},[33,1019,117],{"class":39},[33,1021,1022],{"class":113}," True",[33,1024,1025],{"class":123},"              # ← THIS is the defining difference\n",[33,1027,1028,1031,1033,1035,1037,1039,1041,1044],{"class":35,"line":202},[33,1029,1030],{"class":43},"    you_control: tuple[",[33,1032,181],{"class":113},[33,1034,184],{"class":43},[33,1036,187],{"class":113},[33,1038,190],{"class":43},[33,1040,193],{"class":39},[33,1042,1043],{"class":43}," ()        ",[33,1045,1046],{"class":123},"# nothing — the provider already did this\n",[33,1048,1049],{"class":35,"line":207},[33,1050,78],{"emptyLinePlaceholder":77},[33,1052,1053],{"class":35,"line":219},[33,1054,85],{"class":84},[33,1056,1057,1059,1062],{"class":35,"line":228},[33,1058,91],{"class":39},[33,1060,1061],{"class":84}," InferencePhase",[33,1063,97],{"class":43},[33,1065,1066],{"class":35,"line":237},[33,1067,1068],{"class":103},"    \"\"\"What happens when YOU send a request. You control EVERYTHING here.\"\"\"\n",[33,1070,1071,1073,1075,1077,1079],{"class":35,"line":243},[33,1072,1015],{"class":43},[33,1074,249],{"class":113},[33,1076,117],{"class":39},[33,1078,254],{"class":113},[33,1080,1081],{"class":123},"             # ← frozen. The model learns NOTHING new.\n",[33,1083,1084],{"class":35,"line":263},[33,1085,1086],{"class":123},"    # \"In-context learning\" is a misnomer — no learning occurs. You're selecting\n",[33,1088,1089],{"class":35,"line":273},[33,1090,1091],{"class":123},"    # which pre-trained distribution region to sample from via the tokens you supply.\n",[33,1093,1094,1096,1098,1100,1102,1104,1106],{"class":35,"line":279},[33,1095,1030],{"class":43},[33,1097,181],{"class":113},[33,1099,184],{"class":43},[33,1101,187],{"class":113},[33,1103,190],{"class":43},[33,1105,193],{"class":39},[33,1107,1108],{"class":43}," (\n",[33,1110,1111,1114,1117],{"class":35,"line":285},[33,1112,1113],{"class":103},"        \"system_prompt\"",[33,1115,1116],{"class":43},",       ",[33,1118,1119],{"class":123},"# conditions the distribution toward a role\u002Fbehavior region\n",[33,1121,1122,1125,1128],{"class":35,"line":291},[33,1123,1124],{"class":103},"        \"user_prompt\"",[33,1126,1127],{"class":43},",         ",[33,1129,1130],{"class":123},"# the actual task\n",[33,1132,1133,1136,1139],{"class":35,"line":296},[33,1134,1135],{"class":103},"        \"few_shot_examples\"",[33,1137,1138],{"class":43},",   ",[33,1140,1141],{"class":123},"# demonstrates the input→output mapping in-context\n",[33,1143,1144,1147,1149],{"class":35,"line":308},[33,1145,1146],{"class":103},"        \"temperature\"",[33,1148,1127],{"class":43},[33,1150,1151],{"class":123},"# sampling entropy\n",[33,1153,1154,1157,1159,1162,1165],{"class":35,"line":322},[33,1155,1156],{"class":103},"        \"top_k\"",[33,1158,184],{"class":43},[33,1160,1161],{"class":103},"\"top_p\"",[33,1163,1164],{"class":43},",      ",[33,1166,1167],{"class":123},"# truncation of the candidate token set\n",[33,1169,1170,1173,1176],{"class":35,"line":327},[33,1171,1172],{"class":103},"        \"max_tokens\"",[33,1174,1175],{"class":43},",          ",[33,1177,1178],{"class":123},"# generation budget\n",[33,1180,1181,1184,1186],{"class":35,"line":333},[33,1182,1183],{"class":103},"        \"stop_sequences\"",[33,1185,1164],{"class":43},[33,1187,1188],{"class":123},"# early termination triggers\n",[33,1190,1191],{"class":35,"line":339},[33,1192,1193],{"class":43},"    )\n",[33,1195,1196],{"class":35,"line":345},[33,1197,78],{"emptyLinePlaceholder":77},[33,1199,1200],{"class":35,"line":361},[33,1201,1202],{"class":123},"# --- The key distinction for prompt engineers ---\n",[33,1204,1205],{"class":35,"line":380},[33,1206,1207],{"class":123},"# Training: model.weights \u003C- gradient_step(loss(predictions, targets))\n",[33,1209,1210],{"class":35,"line":385},[33,1211,1212],{"class":123},"#   You weren't there. The model's capability ceiling is already fixed.\n",[33,1214,1215],{"class":35,"line":391},[33,1216,1217],{"class":123},"# Inference: output = sample(model.forward(your_tokens))\n",[33,1219,1220],{"class":35,"line":406},[33,1221,1222],{"class":123},"#   model.weights are READ-ONLY. Your ONLY lever is the token sequence you send.\n",[33,1224,1225],{"class":35,"line":417},[33,1226,1227],{"class":123},"#   A well-crafted few-shot prompt can approximate fine-tuned behavior —\n",[33,1229,1230],{"class":35,"line":440},[33,1231,1232],{"class":123},"#   without touching a single weight. This is why prompting is powerful.\n",[33,1234,1235],{"class":35,"line":467},[33,1236,78],{"emptyLinePlaceholder":77},[33,1238,1239],{"class":35,"line":472},[33,1240,85],{"class":84},[33,1242,1243,1245,1248],{"class":35,"line":478},[33,1244,91],{"class":39},[33,1246,1247],{"class":84}," ChatMemoryIllusion",[33,1249,97],{"class":43},[33,1251,1252],{"class":35,"line":492},[33,1253,1254],{"class":103},"    \"\"\"Explains why 'chat memory' is not real memory.\"\"\"\n",[33,1256,1257],{"class":35,"line":515},[33,1258,1259],{"class":123},"    # Every API call is stateless. The model has NO persistent state between calls.\n",[33,1261,1262],{"class":35,"line":545},[33,1263,1264],{"class":123},"    # What products call \"memory\" is actually:\n",[33,1266,1267],{"class":35,"line":551},[33,1268,1269],{"class":123},"    #   1. Your client code stores prior messages\n",[33,1271,1272],{"class":35,"line":581},[33,1273,1274],{"class":123},"    #   2. On each new request, the ENTIRE transcript is re-sent as input tokens\n",[33,1276,1277],{"class":35,"line":598},[33,1278,1279],{"class":123},"    #   3. The model re-reads everything from scratch — it has no idea it \"remembers\"\n",[33,1281,1282],{"class":35,"line":617},[33,1283,1284],{"class":123},"    # This means: longer \"memory\" = more input tokens = more cost + latency per turn\n",[33,1286,1287],{"class":35,"line":631},[33,1288,1289],{"class":123},"    # AND eventually the transcript exceeds the context window and old messages\n",[33,1291,1292],{"class":35,"line":636},[33,1293,1294],{"class":123},"    # must be dropped, summarized, or truncated — silently, if you're not careful.\n",[33,1296,1297,1300,1302,1304],{"class":35,"line":642},[33,1298,1299],{"class":43},"    api_is_stateless: ",[33,1301,249],{"class":113},[33,1303,117],{"class":39},[33,1305,1306],{"class":113}," True\n",[33,1308,1309,1312,1314,1316],{"class":35,"line":665},[33,1310,1311],{"class":43},"    memory_mechanism: ",[33,1313,181],{"class":113},[33,1315,117],{"class":39},[33,1317,1318],{"class":103}," \"client resends full transcript as tokens each request\"\n",[14,1320,1322],{"id":1321},"tokenization","Tokenization",[19,1324,1326],{"filename":1325,"language":22},"tokenizer_inspection.py",[24,1327,1329],{"className":26,"code":1328,"language":22,"meta":28,"style":28},"\"\"\"\nCompare how different tokenizers split the 'same' text.\nThis is why token-count estimates do NOT transfer between models.\n\"\"\"\nfrom typing import Literal\n\n# --- Simulated BPE (byte-pair encoding) tokenizers for demonstration ---\n# Real tokenizers (tiktoken for GPT, Claude's, Gemini's) have 50K-200K vocab entries.\n# The merge rules are learned from a training corpus — and that corpus is English-heavy.\n\n# In production, use the actual provider tokenizer, never a word-count heuristic:\n#   import tiktoken\n#   enc = tiktoken.encoding_for_model(\"gpt-4o\")\n#   tokens = enc.encode(\"strawberry\")  # → [straw, berry] or similar — NOT ['s','t','r',...]\n\ndef tokenize_naive_word_split(text: str) -> list[str]:\n    \"\"\"What beginners ASSUME the model sees. Wrong — models never see words.\"\"\"\n    return text.split()\n\ndef tokenize_bpe_english(text: str) -> list[str]:\n    \"\"\"Simulated English-trained BPE. Whole words and common subwords merge.\"\"\"\n    # \"strawberry\" might be 1-2 tokens; common words like \"the\" = 1 token\n    # Numbers like \"1234\" might be 1 token (\"1234\") or split (\"12\",\"34\")\n    merges = {\"strawberry\": [\"straw\", \"berry\"], \"tokenization\": [\"token\", \"ization\"],\n              \"the\": [\"the\"], \"1234\": [\"1234\"], \"54321\": [\"543\", \"21\"]}\n    tokens = []\n    for word in text.split():\n        word = word.strip(\",.;!?\")\n        tokens.extend(merges.get(word, [word]))\n    return tokens\n\ndef tokenize_bpe_japanese(text: str) -> list[str]:\n    \"\"\"Simulated BPE applied to Japanese — far less efficient due to English-dominant vocab.\"\"\"\n    # Non-Latin scripts tokenize poorly: same semantic content costs 2-3x more tokens\n    # because the tokenizer's merge table has few non-English entries\n    return list(text.replace(\" \", \"\"))  # char-level fallback — many more tokens\n\n# --- THE strawberry problem ---\n# \"How many r's in strawberry?\" — the model fails because:\n#   1. \"strawberry\" is 1-2 tokens, NOT 9 character tokens\n#   2. The model has no internal character-level representation\n#   3. It must INFER letter composition from training patterns, not read it\n# Fix in production: ask the model to spell it out first, or use a code-execution tool.\n\n# --- THE number tokenization problem ---\n# \"1234\" → [\"1234\"] (1 token)  vs  \"54321\" → [\"543\", \"21\"] (2 tokens)\n# This is why LLM arithmetic is unreliable: multi-digit math is pattern-completion\n# over number-tokens of varying granularity, NOT digit-by-digit school arithmetic.\n# Fix: let the model write intermediate steps (chain-of-thought) or call a calculator tool.\n\n# --- Token budget comparison utility ---\ndef estimate_token_cost(text: str, tokenizer: Literal[\"gpt\", \"claude\", \"gemini\"],\n                        input_price_per_1k: float, output_price_per_1k: float) -> dict:\n    \"\"\"Compare token counts across tokenizers — never assume parity.\"\"\"\n    # In production, call the actual token-count API for each provider.\n    # Heuristic ratios (APPROXIMATE, always verify with real tokenizer):\n    ratios = {\"gpt\": 4.0, \"claude\": 3.8, \"gemini\": 4.2}  # chars per token (English prose)\n    estimated_tokens = max(1, len(text) \u002F\u002F ratios[tokenizer])\n    return {\n        \"tokenizer\": tokenizer,\n        \"char_count\": len(text),\n        \"estimated_tokens\": estimated_tokens,\n        \"estimated_input_cost_usd\": round(estimated_tokens \u002F 1000 * input_price_per_1k, 6),\n        \"warning\": \"Non-English text and code typically cost 1.5-3x more tokens than this estimate\",\n    }\n",[30,1330,1331,1336,1341,1346,1350,1362,1366,1371,1376,1381,1385,1390,1395,1400,1405,1409,1428,1433,1440,1444,1461,1466,1471,1476,1519,1556,1565,1577,1592,1597,1604,1608,1625,1630,1635,1640,1664,1668,1673,1678,1683,1688,1693,1698,1702,1707,1712,1717,1722,1727,1731,1736,1765,1784,1789,1794,1799,1840,1868,1875,1883,1895,1903,1933,1946],{"__ignoreMap":28},[33,1332,1333],{"class":35,"line":36},[33,1334,1335],{"class":103},"\"\"\"\n",[33,1337,1338],{"class":35,"line":47},[33,1339,1340],{"class":103},"Compare how different tokenizers split the 'same' text.\n",[33,1342,1343],{"class":35,"line":61},[33,1344,1345],{"class":103},"This is why token-count estimates do NOT transfer between models.\n",[33,1347,1348],{"class":35,"line":74},[33,1349,1335],{"class":103},[33,1351,1352,1354,1357,1359],{"class":35,"line":81},[33,1353,50],{"class":39},[33,1355,1356],{"class":43}," typing ",[33,1358,40],{"class":39},[33,1360,1361],{"class":43}," Literal\n",[33,1363,1364],{"class":35,"line":88},[33,1365,78],{"emptyLinePlaceholder":77},[33,1367,1368],{"class":35,"line":100},[33,1369,1370],{"class":123},"# --- Simulated BPE (byte-pair encoding) tokenizers for demonstration ---\n",[33,1372,1373],{"class":35,"line":107},[33,1374,1375],{"class":123},"# Real tokenizers (tiktoken for GPT, Claude's, Gemini's) have 50K-200K vocab entries.\n",[33,1377,1378],{"class":35,"line":127},[33,1379,1380],{"class":123},"# The merge rules are learned from a training corpus — and that corpus is English-heavy.\n",[33,1382,1383],{"class":35,"line":144},[33,1384,78],{"emptyLinePlaceholder":77},[33,1386,1387],{"class":35,"line":159},[33,1388,1389],{"class":123},"# In production, use the actual provider tokenizer, never a word-count heuristic:\n",[33,1391,1392],{"class":35,"line":175},[33,1393,1394],{"class":123},"#   import tiktoken\n",[33,1396,1397],{"class":35,"line":202},[33,1398,1399],{"class":123},"#   enc = tiktoken.encoding_for_model(\"gpt-4o\")\n",[33,1401,1402],{"class":35,"line":207},[33,1403,1404],{"class":123},"#   tokens = enc.encode(\"strawberry\")  # → [straw, berry] or similar — NOT ['s','t','r',...]\n",[33,1406,1407],{"class":35,"line":219},[33,1408,78],{"emptyLinePlaceholder":77},[33,1410,1411,1413,1416,1419,1421,1424,1426],{"class":35,"line":228},[33,1412,210],{"class":39},[33,1414,1415],{"class":84}," tokenize_naive_word_split",[33,1417,1418],{"class":43},"(text: ",[33,1420,181],{"class":113},[33,1422,1423],{"class":43},") -> list[",[33,1425,181],{"class":113},[33,1427,747],{"class":43},[33,1429,1430],{"class":35,"line":237},[33,1431,1432],{"class":103},"    \"\"\"What beginners ASSUME the model sees. Wrong — models never see words.\"\"\"\n",[33,1434,1435,1437],{"class":35,"line":243},[33,1436,718],{"class":39},[33,1438,1439],{"class":43}," text.split()\n",[33,1441,1442],{"class":35,"line":263},[33,1443,78],{"emptyLinePlaceholder":77},[33,1445,1446,1448,1451,1453,1455,1457,1459],{"class":35,"line":273},[33,1447,210],{"class":39},[33,1449,1450],{"class":84}," tokenize_bpe_english",[33,1452,1418],{"class":43},[33,1454,181],{"class":113},[33,1456,1423],{"class":43},[33,1458,181],{"class":113},[33,1460,747],{"class":43},[33,1462,1463],{"class":35,"line":279},[33,1464,1465],{"class":103},"    \"\"\"Simulated English-trained BPE. Whole words and common subwords merge.\"\"\"\n",[33,1467,1468],{"class":35,"line":285},[33,1469,1470],{"class":123},"    # \"strawberry\" might be 1-2 tokens; common words like \"the\" = 1 token\n",[33,1472,1473],{"class":35,"line":291},[33,1474,1475],{"class":123},"    # Numbers like \"1234\" might be 1 token (\"1234\") or split (\"12\",\"34\")\n",[33,1477,1478,1481,1483,1486,1489,1492,1495,1497,1500,1503,1506,1508,1511,1513,1516],{"class":35,"line":296},[33,1479,1480],{"class":43},"    merges ",[33,1482,193],{"class":39},[33,1484,1485],{"class":43}," {",[33,1487,1488],{"class":103},"\"strawberry\"",[33,1490,1491],{"class":43},": [",[33,1493,1494],{"class":103},"\"straw\"",[33,1496,184],{"class":43},[33,1498,1499],{"class":103},"\"berry\"",[33,1501,1502],{"class":43},"], ",[33,1504,1505],{"class":103},"\"tokenization\"",[33,1507,1491],{"class":43},[33,1509,1510],{"class":103},"\"token\"",[33,1512,184],{"class":43},[33,1514,1515],{"class":103},"\"ization\"",[33,1517,1518],{"class":43},"],\n",[33,1520,1521,1524,1526,1529,1531,1534,1536,1538,1540,1543,1545,1548,1550,1553],{"class":35,"line":308},[33,1522,1523],{"class":103},"              \"the\"",[33,1525,1491],{"class":43},[33,1527,1528],{"class":103},"\"the\"",[33,1530,1502],{"class":43},[33,1532,1533],{"class":103},"\"1234\"",[33,1535,1491],{"class":43},[33,1537,1533],{"class":103},[33,1539,1502],{"class":43},[33,1541,1542],{"class":103},"\"54321\"",[33,1544,1491],{"class":43},[33,1546,1547],{"class":103},"\"543\"",[33,1549,184],{"class":43},[33,1551,1552],{"class":103},"\"21\"",[33,1554,1555],{"class":43},"]}\n",[33,1557,1558,1561,1563],{"class":35,"line":322},[33,1559,1560],{"class":43},"    tokens ",[33,1562,193],{"class":39},[33,1564,779],{"class":43},[33,1566,1567,1569,1572,1574],{"class":35,"line":327},[33,1568,785],{"class":39},[33,1570,1571],{"class":43}," word ",[33,1573,791],{"class":39},[33,1575,1576],{"class":43}," text.split():\n",[33,1578,1579,1582,1584,1587,1590],{"class":35,"line":333},[33,1580,1581],{"class":43},"        word ",[33,1583,193],{"class":39},[33,1585,1586],{"class":43}," word.strip(",[33,1588,1589],{"class":103},"\",.;!?\"",[33,1591,512],{"class":43},[33,1593,1594],{"class":35,"line":339},[33,1595,1596],{"class":43},"        tokens.extend(merges.get(word, [word]))\n",[33,1598,1599,1601],{"class":35,"line":345},[33,1600,718],{"class":39},[33,1602,1603],{"class":43}," tokens\n",[33,1605,1606],{"class":35,"line":361},[33,1607,78],{"emptyLinePlaceholder":77},[33,1609,1610,1612,1615,1617,1619,1621,1623],{"class":35,"line":380},[33,1611,210],{"class":39},[33,1613,1614],{"class":84}," tokenize_bpe_japanese",[33,1616,1418],{"class":43},[33,1618,181],{"class":113},[33,1620,1423],{"class":43},[33,1622,181],{"class":113},[33,1624,747],{"class":43},[33,1626,1627],{"class":35,"line":385},[33,1628,1629],{"class":103},"    \"\"\"Simulated BPE applied to Japanese — far less efficient due to English-dominant vocab.\"\"\"\n",[33,1631,1632],{"class":35,"line":391},[33,1633,1634],{"class":123},"    # Non-Latin scripts tokenize poorly: same semantic content costs 2-3x more tokens\n",[33,1636,1637],{"class":35,"line":406},[33,1638,1639],{"class":123},"    # because the tokenizer's merge table has few non-English entries\n",[33,1641,1642,1644,1647,1650,1653,1655,1658,1661],{"class":35,"line":417},[33,1643,718],{"class":39},[33,1645,1646],{"class":113}," list",[33,1648,1649],{"class":43},"(text.replace(",[33,1651,1652],{"class":103},"\" \"",[33,1654,184],{"class":43},[33,1656,1657],{"class":103},"\"\"",[33,1659,1660],{"class":43},"))  ",[33,1662,1663],{"class":123},"# char-level fallback — many more tokens\n",[33,1665,1666],{"class":35,"line":440},[33,1667,78],{"emptyLinePlaceholder":77},[33,1669,1670],{"class":35,"line":467},[33,1671,1672],{"class":123},"# --- THE strawberry problem ---\n",[33,1674,1675],{"class":35,"line":472},[33,1676,1677],{"class":123},"# \"How many r's in strawberry?\" — the model fails because:\n",[33,1679,1680],{"class":35,"line":478},[33,1681,1682],{"class":123},"#   1. \"strawberry\" is 1-2 tokens, NOT 9 character tokens\n",[33,1684,1685],{"class":35,"line":492},[33,1686,1687],{"class":123},"#   2. The model has no internal character-level representation\n",[33,1689,1690],{"class":35,"line":515},[33,1691,1692],{"class":123},"#   3. It must INFER letter composition from training patterns, not read it\n",[33,1694,1695],{"class":35,"line":545},[33,1696,1697],{"class":123},"# Fix in production: ask the model to spell it out first, or use a code-execution tool.\n",[33,1699,1700],{"class":35,"line":551},[33,1701,78],{"emptyLinePlaceholder":77},[33,1703,1704],{"class":35,"line":581},[33,1705,1706],{"class":123},"# --- THE number tokenization problem ---\n",[33,1708,1709],{"class":35,"line":598},[33,1710,1711],{"class":123},"# \"1234\" → [\"1234\"] (1 token)  vs  \"54321\" → [\"543\", \"21\"] (2 tokens)\n",[33,1713,1714],{"class":35,"line":617},[33,1715,1716],{"class":123},"# This is why LLM arithmetic is unreliable: multi-digit math is pattern-completion\n",[33,1718,1719],{"class":35,"line":631},[33,1720,1721],{"class":123},"# over number-tokens of varying granularity, NOT digit-by-digit school arithmetic.\n",[33,1723,1724],{"class":35,"line":636},[33,1725,1726],{"class":123},"# Fix: let the model write intermediate steps (chain-of-thought) or call a calculator tool.\n",[33,1728,1729],{"class":35,"line":642},[33,1730,78],{"emptyLinePlaceholder":77},[33,1732,1733],{"class":35,"line":665},[33,1734,1735],{"class":123},"# --- Token budget comparison utility ---\n",[33,1737,1738,1740,1743,1745,1747,1750,1753,1755,1758,1760,1763],{"class":35,"line":686},[33,1739,210],{"class":39},[33,1741,1742],{"class":84}," estimate_token_cost",[33,1744,1418],{"class":43},[33,1746,181],{"class":113},[33,1748,1749],{"class":43},", tokenizer: Literal[",[33,1751,1752],{"class":103},"\"gpt\"",[33,1754,184],{"class":43},[33,1756,1757],{"class":103},"\"claude\"",[33,1759,184],{"class":43},[33,1761,1762],{"class":103},"\"gemini\"",[33,1764,1518],{"class":43},[33,1766,1767,1770,1772,1775,1777,1779,1782],{"class":35,"line":691},[33,1768,1769],{"class":43},"                        input_price_per_1k: ",[33,1771,114],{"class":113},[33,1773,1774],{"class":43},", output_price_per_1k: ",[33,1776,114],{"class":113},[33,1778,266],{"class":43},[33,1780,1781],{"class":113},"dict",[33,1783,97],{"class":43},[33,1785,1786],{"class":35,"line":703},[33,1787,1788],{"class":103},"    \"\"\"Compare token counts across tokenizers — never assume parity.\"\"\"\n",[33,1790,1791],{"class":35,"line":709},[33,1792,1793],{"class":123},"    # In production, call the actual token-count API for each provider.\n",[33,1795,1796],{"class":35,"line":715},[33,1797,1798],{"class":123},"    # Heuristic ratios (APPROXIMATE, always verify with real tokenizer):\n",[33,1800,1801,1804,1806,1808,1810,1813,1816,1818,1820,1822,1825,1827,1829,1831,1834,1837],{"class":35,"line":724},[33,1802,1803],{"class":43},"    ratios ",[33,1805,193],{"class":39},[33,1807,1485],{"class":43},[33,1809,1752],{"class":103},[33,1811,1812],{"class":43},": ",[33,1814,1815],{"class":113},"4.0",[33,1817,184],{"class":43},[33,1819,1757],{"class":103},[33,1821,1812],{"class":43},[33,1823,1824],{"class":113},"3.8",[33,1826,184],{"class":43},[33,1828,1762],{"class":103},[33,1830,1812],{"class":43},[33,1832,1833],{"class":113},"4.2",[33,1835,1836],{"class":43},"}  ",[33,1838,1839],{"class":123},"# chars per token (English prose)\n",[33,1841,1842,1845,1847,1850,1852,1854,1856,1859,1862,1865],{"class":35,"line":729},[33,1843,1844],{"class":43},"    estimated_tokens ",[33,1846,193],{"class":39},[33,1848,1849],{"class":113}," max",[33,1851,458],{"class":43},[33,1853,431],{"class":113},[33,1855,184],{"class":43},[33,1857,1858],{"class":113},"len",[33,1860,1861],{"class":43},"(text) ",[33,1863,1864],{"class":39},"\u002F\u002F",[33,1866,1867],{"class":43}," ratios[tokenizer])\n",[33,1869,1870,1872],{"class":35,"line":750},[33,1871,718],{"class":39},[33,1873,1874],{"class":43}," {\n",[33,1876,1877,1880],{"class":35,"line":756},[33,1878,1879],{"class":103},"        \"tokenizer\"",[33,1881,1882],{"class":43},": tokenizer,\n",[33,1884,1885,1888,1890,1892],{"class":35,"line":767},[33,1886,1887],{"class":103},"        \"char_count\"",[33,1889,1812],{"class":43},[33,1891,1858],{"class":113},[33,1893,1894],{"class":43},"(text),\n",[33,1896,1897,1900],{"class":35,"line":782},[33,1898,1899],{"class":103},"        \"estimated_tokens\"",[33,1901,1902],{"class":43},": estimated_tokens,\n",[33,1904,1905,1908,1910,1913,1916,1918,1921,1924,1927,1930],{"class":35,"line":800},[33,1906,1907],{"class":103},"        \"estimated_input_cost_usd\"",[33,1909,1812],{"class":43},[33,1911,1912],{"class":113},"round",[33,1914,1915],{"class":43},"(estimated_tokens ",[33,1917,371],{"class":39},[33,1919,1920],{"class":113}," 1000",[33,1922,1923],{"class":39}," *",[33,1925,1926],{"class":43}," input_price_per_1k, ",[33,1928,1929],{"class":113},"6",[33,1931,1932],{"class":43},"),\n",[33,1934,1935,1938,1940,1943],{"class":35,"line":826},[33,1936,1937],{"class":103},"        \"warning\"",[33,1939,1812],{"class":43},[33,1941,1942],{"class":103},"\"Non-English text and code typically cost 1.5-3x more tokens than this estimate\"",[33,1944,1945],{"class":43},",\n",[33,1947,1948],{"class":35,"line":843},[33,1949,1950],{"class":43},"    }\n",[14,1952,1954],{"id":1953},"context-windows","Context Windows",[19,1956,1958],{"filename":1957,"language":22},"context_budget_allocator.py",[24,1959,1961],{"className":26,"code":1960,"language":22,"meta":28,"style":28},"\"\"\"\nA production context-budget allocator.\nContext window is a BUDGET, not a bottomless bucket — every token costs money + latency.\n\"\"\"\nfrom dataclasses import dataclass, field\nfrom enum import Enum\n\nclass TokenAllocationError(Exception):\n    \"\"\"Raised when allocations exceed the context budget — fail loudly, never truncate silently.\"\"\"\n\nclass Section(Enum):\n    SYSTEM_PROMPT    = \"system_prompt\"\n    FEW_SHOT         = \"few_shot_examples\"\n    USER_MESSAGE     = \"user_message\"\n    RETRIEVED_DOCS   = \"retrieved_documents\"\n    CONVERSATION     = \"conversation_history\"\n    TOOL_OUTPUTS     = \"tool_outputs\"\n    RESPONSE_RESERVE = \"response_reserve\"   # space set aside for the model's output\n\n@dataclass\nclass ContextBudget:\n    \"\"\"Validated token budget for a single API request.\"\"\"\n    model: str\n    max_context: int                              # provider-documented context window limit\n    allocations: dict[Section, int] = field(default_factory=dict)\n    reserved_for_response: int = 4096             # never let input eat the response space\n\n    def __post_init__(self):\n        if self.reserved_for_response >= self.max_context:\n            raise TokenAllocationError(\n                f\"Response reserve ({self.reserved_for_response}) must be \u003C context window ({self.max_context})\"\n            )\n        self.allocations[Section.RESPONSE_RESERVE] = self.reserved_for_response\n\n    @property\n    def available_for_input(self) -> int:\n        \"\"\"The hard ceiling for all input tokens (everything except the model's response).\"\"\"\n        return self.max_context - self.reserved_for_response\n\n    @property\n    def total_allocated(self) -> int:\n        return sum(v for k, v in self.allocations.items() if k != Section.RESPONSE_RESERVE)\n\n    @property\n    def remaining(self) -> int:\n        return self.available_for_input - self.total_allocated\n\n    def allocate(self, section: Section, tokens: int, strict: bool = True) -> None:\n        \"\"\"Reserve tokens for a section. Raises if it would overflow.\"\"\"\n        if tokens \u003C 0:\n            raise TokenAllocationError(f\"Cannot allocate negative tokens for {section.value}\")\n        tentative_total = self.total_allocated + tokens - self.allocations.get(section, 0)\n        if strict and tentative_total > self.available_for_input:\n            raise TokenAllocationError(\n                f\"Allocating {tokens} tokens to {section.value} would exceed input budget \"\n                f\"({tentative_total}\u002F{self.available_for_input}). \"\n                f\"Currently allocated: {self.total_allocated}. \"\n                f\"Overflow by {tentative_total - self.available_for_input} tokens.\"\n            )\n        self.allocations[section] = tokens\n\n    def truncate_to_fit(self, section: Section, content_tokens: int) -> int:\n        \"\"\"How many tokens of `content_tokens` can actually fit in the remaining budget.\"\"\"\n        space = min(content_tokens, self.available_for_input - self.total_allocated)\n        return max(0, space)\n\n# --- Production usage ---\nbudget = ContextBudget(model=\"claude-sonnet\", max_context=200_000, reserved_for_response=4096)\nbudget.allocate(Section.SYSTEM_PROMPT, 800)\nbudget.allocate(Section.FEW_SHOT, 1_200)\nbudget.allocate(Section.RETRIEVED_DOCS, 50_000)\nbudget.allocate(Section.CONVERSATION, 30_000)\n\n# Simulate a user pasting a massive document\ndoc_tokens = 150_000\nfits = budget.truncate_to_fit(Section.USER_MESSAGE, doc_tokens)\n# → only ~118,000 tokens fit — the rest MUST be dropped, summarized, or chunked.\n# NEVER silently truncate: the caller decides the strategy (summarize? embed+retrieve? error?)\n\n# --- THE \"lost in the middle\" effect ---\n# A 200K context window means the API ACCEPTS 200K tokens — NOT that the model\n# reliably USES all 200K. Attention weight is position-dependent:\n#   - Beginning tokens: high recall (primacy effect)\n#   - End tokens: high recall (recency effect)\n#   - Middle tokens: degraded recall (the \"lost in the middle\" valley)\n# Production implication: put critical instructions at the START or END of context,\n# not buried in the middle of a 50-page document dump.\n",[30,1962,1963,1967,1972,1977,1981,1992,2004,2008,2023,2028,2032,2046,2057,2068,2079,2090,2100,2110,2123,2127,2131,2140,2145,2153,2163,2186,2201,2205,2216,2234,2242,2272,2277,2297,2301,2309,2323,2328,2344,2348,2354,2367,2405,2409,2415,2428,2444,2448,2476,2481,2494,2520,2549,2569,2575,2601,2627,2644,2667,2671,2682,2686,2704,2709,2734,2747,2751,2756,2796,2811,2825,2839,2854,2859,2865,2876,2893,2899,2905,2910,2916,2922,2928,2934,2940,2946,2952],{"__ignoreMap":28},[33,1964,1965],{"class":35,"line":36},[33,1966,1335],{"class":103},[33,1968,1969],{"class":35,"line":47},[33,1970,1971],{"class":103},"A production context-budget allocator.\n",[33,1973,1974],{"class":35,"line":61},[33,1975,1976],{"class":103},"Context window is a BUDGET, not a bottomless bucket — every token costs money + latency.\n",[33,1978,1979],{"class":35,"line":74},[33,1980,1335],{"class":103},[33,1982,1983,1985,1987,1989],{"class":35,"line":81},[33,1984,50],{"class":39},[33,1986,66],{"class":43},[33,1988,40],{"class":39},[33,1990,1991],{"class":43}," dataclass, field\n",[33,1993,1994,1996,1999,2001],{"class":35,"line":88},[33,1995,50],{"class":39},[33,1997,1998],{"class":43}," enum ",[33,2000,40],{"class":39},[33,2002,2003],{"class":43}," Enum\n",[33,2005,2006],{"class":35,"line":100},[33,2007,78],{"emptyLinePlaceholder":77},[33,2009,2010,2012,2015,2017,2020],{"class":35,"line":107},[33,2011,91],{"class":39},[33,2013,2014],{"class":84}," TokenAllocationError",[33,2016,458],{"class":43},[33,2018,2019],{"class":113},"Exception",[33,2021,2022],{"class":43},"):\n",[33,2024,2025],{"class":35,"line":127},[33,2026,2027],{"class":103},"    \"\"\"Raised when allocations exceed the context budget — fail loudly, never truncate silently.\"\"\"\n",[33,2029,2030],{"class":35,"line":144},[33,2031,78],{"emptyLinePlaceholder":77},[33,2033,2034,2036,2039,2041,2044],{"class":35,"line":159},[33,2035,91],{"class":39},[33,2037,2038],{"class":84}," Section",[33,2040,458],{"class":43},[33,2042,2043],{"class":84},"Enum",[33,2045,2022],{"class":43},[33,2047,2048,2051,2054],{"class":35,"line":175},[33,2049,2050],{"class":113},"    SYSTEM_PROMPT",[33,2052,2053],{"class":39},"    =",[33,2055,2056],{"class":103}," \"system_prompt\"\n",[33,2058,2059,2062,2065],{"class":35,"line":202},[33,2060,2061],{"class":113},"    FEW_SHOT",[33,2063,2064],{"class":39},"         =",[33,2066,2067],{"class":103}," \"few_shot_examples\"\n",[33,2069,2070,2073,2076],{"class":35,"line":207},[33,2071,2072],{"class":113},"    USER_MESSAGE",[33,2074,2075],{"class":39},"     =",[33,2077,2078],{"class":103}," \"user_message\"\n",[33,2080,2081,2084,2087],{"class":35,"line":219},[33,2082,2083],{"class":113},"    RETRIEVED_DOCS",[33,2085,2086],{"class":39},"   =",[33,2088,2089],{"class":103}," \"retrieved_documents\"\n",[33,2091,2092,2095,2097],{"class":35,"line":228},[33,2093,2094],{"class":113},"    CONVERSATION",[33,2096,2075],{"class":39},[33,2098,2099],{"class":103}," \"conversation_history\"\n",[33,2101,2102,2105,2107],{"class":35,"line":237},[33,2103,2104],{"class":113},"    TOOL_OUTPUTS",[33,2106,2075],{"class":39},[33,2108,2109],{"class":103}," \"tool_outputs\"\n",[33,2111,2112,2115,2117,2120],{"class":35,"line":243},[33,2113,2114],{"class":113},"    RESPONSE_RESERVE",[33,2116,117],{"class":39},[33,2118,2119],{"class":103}," \"response_reserve\"",[33,2121,2122],{"class":123},"   # space set aside for the model's output\n",[33,2124,2125],{"class":35,"line":263},[33,2126,78],{"emptyLinePlaceholder":77},[33,2128,2129],{"class":35,"line":273},[33,2130,85],{"class":84},[33,2132,2133,2135,2138],{"class":35,"line":279},[33,2134,91],{"class":39},[33,2136,2137],{"class":84}," ContextBudget",[33,2139,97],{"class":43},[33,2141,2142],{"class":35,"line":285},[33,2143,2144],{"class":103},"    \"\"\"Validated token budget for a single API request.\"\"\"\n",[33,2146,2147,2150],{"class":35,"line":291},[33,2148,2149],{"class":43},"    model: ",[33,2151,2152],{"class":113},"str\n",[33,2154,2155,2158,2160],{"class":35,"line":296},[33,2156,2157],{"class":43},"    max_context: ",[33,2159,133],{"class":113},[33,2161,2162],{"class":123},"                              # provider-documented context window limit\n",[33,2164,2165,2168,2170,2172,2174,2177,2180,2182,2184],{"class":35,"line":308},[33,2166,2167],{"class":43},"    allocations: dict[Section, ",[33,2169,133],{"class":113},[33,2171,190],{"class":43},[33,2173,193],{"class":39},[33,2175,2176],{"class":43}," field(",[33,2178,2179],{"class":503},"default_factory",[33,2181,193],{"class":39},[33,2183,1781],{"class":113},[33,2185,512],{"class":43},[33,2187,2188,2191,2193,2195,2198],{"class":35,"line":322},[33,2189,2190],{"class":43},"    reserved_for_response: ",[33,2192,133],{"class":113},[33,2194,117],{"class":39},[33,2196,2197],{"class":113}," 4096",[33,2199,2200],{"class":123},"             # never let input eat the response space\n",[33,2202,2203],{"class":35,"line":327},[33,2204,78],{"emptyLinePlaceholder":77},[33,2206,2207,2210,2213],{"class":35,"line":333},[33,2208,2209],{"class":39},"    def",[33,2211,2212],{"class":113}," __post_init__",[33,2214,2215],{"class":43},"(self):\n",[33,2217,2218,2220,2223,2226,2229,2231],{"class":35,"line":339},[33,2219,829],{"class":39},[33,2221,2222],{"class":113}," self",[33,2224,2225],{"class":43},".reserved_for_response ",[33,2227,2228],{"class":39},">=",[33,2230,2222],{"class":113},[33,2232,2233],{"class":43},".max_context:\n",[33,2235,2236,2239],{"class":35,"line":345},[33,2237,2238],{"class":39},"            raise",[33,2240,2241],{"class":43}," TokenAllocationError(\n",[33,2243,2244,2247,2250,2253,2256,2259,2262,2264,2267,2269],{"class":35,"line":361},[33,2245,2246],{"class":39},"                f",[33,2248,2249],{"class":103},"\"Response reserve (",[33,2251,2252],{"class":113},"{self",[33,2254,2255],{"class":43},".reserved_for_response",[33,2257,2258],{"class":113},"}",[33,2260,2261],{"class":103},") must be \u003C context window (",[33,2263,2252],{"class":113},[33,2265,2266],{"class":43},".max_context",[33,2268,2258],{"class":113},[33,2270,2271],{"class":103},")\"\n",[33,2273,2274],{"class":35,"line":380},[33,2275,2276],{"class":43},"            )\n",[33,2278,2279,2282,2285,2288,2290,2292,2294],{"class":35,"line":385},[33,2280,2281],{"class":113},"        self",[33,2283,2284],{"class":43},".allocations[Section.",[33,2286,2287],{"class":113},"RESPONSE_RESERVE",[33,2289,190],{"class":43},[33,2291,193],{"class":39},[33,2293,2222],{"class":113},[33,2295,2296],{"class":43},".reserved_for_response\n",[33,2298,2299],{"class":35,"line":391},[33,2300,78],{"emptyLinePlaceholder":77},[33,2302,2303,2306],{"class":35,"line":406},[33,2304,2305],{"class":84},"    @",[33,2307,2308],{"class":113},"property\n",[33,2310,2311,2313,2316,2319,2321],{"class":35,"line":417},[33,2312,2209],{"class":39},[33,2314,2315],{"class":84}," available_for_input",[33,2317,2318],{"class":43},"(self) -> ",[33,2320,133],{"class":113},[33,2322,97],{"class":43},[33,2324,2325],{"class":35,"line":440},[33,2326,2327],{"class":103},"        \"\"\"The hard ceiling for all input tokens (everything except the model's response).\"\"\"\n",[33,2329,2330,2333,2335,2338,2340,2342],{"class":35,"line":467},[33,2331,2332],{"class":39},"        return",[33,2334,2222],{"class":113},[33,2336,2337],{"class":43},".max_context ",[33,2339,428],{"class":39},[33,2341,2222],{"class":113},[33,2343,2296],{"class":43},[33,2345,2346],{"class":35,"line":472},[33,2347,78],{"emptyLinePlaceholder":77},[33,2349,2350,2352],{"class":35,"line":478},[33,2351,2305],{"class":84},[33,2353,2308],{"class":113},[33,2355,2356,2358,2361,2363,2365],{"class":35,"line":492},[33,2357,2209],{"class":39},[33,2359,2360],{"class":84}," total_allocated",[33,2362,2318],{"class":43},[33,2364,133],{"class":113},[33,2366,97],{"class":43},[33,2368,2369,2371,2374,2377,2380,2383,2385,2387,2390,2393,2396,2398,2401,2403],{"class":35,"line":515},[33,2370,2332],{"class":39},[33,2372,2373],{"class":113}," sum",[33,2375,2376],{"class":43},"(v ",[33,2378,2379],{"class":39},"for",[33,2381,2382],{"class":43}," k, v ",[33,2384,791],{"class":39},[33,2386,2222],{"class":113},[33,2388,2389],{"class":43},".allocations.items() ",[33,2391,2392],{"class":39},"if",[33,2394,2395],{"class":43}," k ",[33,2397,354],{"class":39},[33,2399,2400],{"class":43}," Section.",[33,2402,2287],{"class":113},[33,2404,512],{"class":43},[33,2406,2407],{"class":35,"line":545},[33,2408,78],{"emptyLinePlaceholder":77},[33,2410,2411,2413],{"class":35,"line":551},[33,2412,2305],{"class":84},[33,2414,2308],{"class":113},[33,2416,2417,2419,2422,2424,2426],{"class":35,"line":581},[33,2418,2209],{"class":39},[33,2420,2421],{"class":84}," remaining",[33,2423,2318],{"class":43},[33,2425,133],{"class":113},[33,2427,97],{"class":43},[33,2429,2430,2432,2434,2437,2439,2441],{"class":35,"line":598},[33,2431,2332],{"class":39},[33,2433,2222],{"class":113},[33,2435,2436],{"class":43},".available_for_input ",[33,2438,428],{"class":39},[33,2440,2222],{"class":113},[33,2442,2443],{"class":43},".total_allocated\n",[33,2445,2446],{"class":35,"line":617},[33,2447,78],{"emptyLinePlaceholder":77},[33,2449,2450,2452,2455,2458,2460,2463,2465,2467,2469,2471,2474],{"class":35,"line":631},[33,2451,2209],{"class":39},[33,2453,2454],{"class":84}," allocate",[33,2456,2457],{"class":43},"(self, section: Section, tokens: ",[33,2459,133],{"class":113},[33,2461,2462],{"class":43},", strict: ",[33,2464,249],{"class":113},[33,2466,117],{"class":39},[33,2468,1022],{"class":113},[33,2470,266],{"class":43},[33,2472,2473],{"class":113},"None",[33,2475,97],{"class":43},[33,2477,2478],{"class":35,"line":636},[33,2479,2480],{"class":103},"        \"\"\"Reserve tokens for a section. Raises if it would overflow.\"\"\"\n",[33,2482,2483,2485,2488,2490,2492],{"class":35,"line":642},[33,2484,829],{"class":39},[33,2486,2487],{"class":43}," tokens ",[33,2489,450],{"class":39},[33,2491,138],{"class":113},[33,2493,97],{"class":43},[33,2495,2496,2498,2501,2504,2507,2510,2513,2515,2518],{"class":35,"line":665},[33,2497,2238],{"class":39},[33,2499,2500],{"class":43}," TokenAllocationError(",[33,2502,2503],{"class":39},"f",[33,2505,2506],{"class":103},"\"Cannot allocate negative tokens for ",[33,2508,2509],{"class":113},"{",[33,2511,2512],{"class":43},"section.value",[33,2514,2258],{"class":113},[33,2516,2517],{"class":103},"\"",[33,2519,512],{"class":43},[33,2521,2522,2525,2527,2529,2532,2535,2537,2539,2541,2544,2547],{"class":35,"line":686},[33,2523,2524],{"class":43},"        tentative_total ",[33,2526,193],{"class":39},[33,2528,2222],{"class":113},[33,2530,2531],{"class":43},".total_allocated ",[33,2533,2534],{"class":39},"+",[33,2536,2487],{"class":43},[33,2538,428],{"class":39},[33,2540,2222],{"class":113},[33,2542,2543],{"class":43},".allocations.get(section, ",[33,2545,2546],{"class":113},"0",[33,2548,512],{"class":43},[33,2550,2551,2553,2556,2559,2562,2564,2566],{"class":35,"line":691},[33,2552,829],{"class":39},[33,2554,2555],{"class":43}," strict ",[33,2557,2558],{"class":39},"and",[33,2560,2561],{"class":43}," tentative_total ",[33,2563,399],{"class":39},[33,2565,2222],{"class":113},[33,2567,2568],{"class":43},".available_for_input:\n",[33,2570,2571,2573],{"class":35,"line":703},[33,2572,2238],{"class":39},[33,2574,2241],{"class":43},[33,2576,2577,2579,2582,2584,2587,2589,2592,2594,2596,2598],{"class":35,"line":709},[33,2578,2246],{"class":39},[33,2580,2581],{"class":103},"\"Allocating ",[33,2583,2509],{"class":113},[33,2585,2586],{"class":43},"tokens",[33,2588,2258],{"class":113},[33,2590,2591],{"class":103}," tokens to ",[33,2593,2509],{"class":113},[33,2595,2512],{"class":43},[33,2597,2258],{"class":113},[33,2599,2600],{"class":103}," would exceed input budget \"\n",[33,2602,2603,2605,2608,2610,2613,2615,2617,2619,2622,2624],{"class":35,"line":715},[33,2604,2246],{"class":39},[33,2606,2607],{"class":103},"\"(",[33,2609,2509],{"class":113},[33,2611,2612],{"class":43},"tentative_total",[33,2614,2258],{"class":113},[33,2616,371],{"class":103},[33,2618,2252],{"class":113},[33,2620,2621],{"class":43},".available_for_input",[33,2623,2258],{"class":113},[33,2625,2626],{"class":103},"). \"\n",[33,2628,2629,2631,2634,2636,2639,2641],{"class":35,"line":724},[33,2630,2246],{"class":39},[33,2632,2633],{"class":103},"\"Currently allocated: ",[33,2635,2252],{"class":113},[33,2637,2638],{"class":43},".total_allocated",[33,2640,2258],{"class":113},[33,2642,2643],{"class":103},". \"\n",[33,2645,2646,2648,2651,2653,2656,2658,2660,2662,2664],{"class":35,"line":729},[33,2647,2246],{"class":39},[33,2649,2650],{"class":103},"\"Overflow by ",[33,2652,2509],{"class":113},[33,2654,2655],{"class":43},"tentative_total ",[33,2657,428],{"class":39},[33,2659,2222],{"class":113},[33,2661,2621],{"class":43},[33,2663,2258],{"class":113},[33,2665,2666],{"class":103}," tokens.\"\n",[33,2668,2669],{"class":35,"line":750},[33,2670,2276],{"class":43},[33,2672,2673,2675,2678,2680],{"class":35,"line":756},[33,2674,2281],{"class":113},[33,2676,2677],{"class":43},".allocations[section] ",[33,2679,193],{"class":39},[33,2681,1603],{"class":43},[33,2683,2684],{"class":35,"line":767},[33,2685,78],{"emptyLinePlaceholder":77},[33,2687,2688,2690,2693,2696,2698,2700,2702],{"class":35,"line":782},[33,2689,2209],{"class":39},[33,2691,2692],{"class":84}," truncate_to_fit",[33,2694,2695],{"class":43},"(self, section: Section, content_tokens: ",[33,2697,133],{"class":113},[33,2699,266],{"class":43},[33,2701,133],{"class":113},[33,2703,97],{"class":43},[33,2705,2706],{"class":35,"line":800},[33,2707,2708],{"class":103},"        \"\"\"How many tokens of `content_tokens` can actually fit in the remaining budget.\"\"\"\n",[33,2710,2711,2714,2716,2719,2722,2725,2727,2729,2731],{"class":35,"line":826},[33,2712,2713],{"class":43},"        space ",[33,2715,193],{"class":39},[33,2717,2718],{"class":113}," min",[33,2720,2721],{"class":43},"(content_tokens, ",[33,2723,2724],{"class":113},"self",[33,2726,2436],{"class":43},[33,2728,428],{"class":39},[33,2730,2222],{"class":113},[33,2732,2733],{"class":43},".total_allocated)\n",[33,2735,2736,2738,2740,2742,2744],{"class":35,"line":843},[33,2737,2332],{"class":39},[33,2739,1849],{"class":113},[33,2741,458],{"class":43},[33,2743,2546],{"class":113},[33,2745,2746],{"class":43},", space)\n",[33,2748,2749],{"class":35,"line":849},[33,2750,78],{"emptyLinePlaceholder":77},[33,2752,2753],{"class":35,"line":855},[33,2754,2755],{"class":123},"# --- Production usage ---\n",[33,2757,2758,2761,2763,2766,2769,2771,2774,2776,2779,2781,2784,2786,2789,2791,2794],{"class":35,"line":866},[33,2759,2760],{"class":43},"budget ",[33,2762,193],{"class":39},[33,2764,2765],{"class":43}," ContextBudget(",[33,2767,2768],{"class":503},"model",[33,2770,193],{"class":39},[33,2772,2773],{"class":103},"\"claude-sonnet\"",[33,2775,184],{"class":43},[33,2777,2778],{"class":503},"max_context",[33,2780,193],{"class":39},[33,2782,2783],{"class":113},"200_000",[33,2785,184],{"class":43},[33,2787,2788],{"class":503},"reserved_for_response",[33,2790,193],{"class":39},[33,2792,2793],{"class":113},"4096",[33,2795,512],{"class":43},[33,2797,2798,2801,2804,2806,2809],{"class":35,"line":872},[33,2799,2800],{"class":43},"budget.allocate(Section.",[33,2802,2803],{"class":113},"SYSTEM_PROMPT",[33,2805,184],{"class":43},[33,2807,2808],{"class":113},"800",[33,2810,512],{"class":43},[33,2812,2813,2815,2818,2820,2823],{"class":35,"line":878},[33,2814,2800],{"class":43},[33,2816,2817],{"class":113},"FEW_SHOT",[33,2819,184],{"class":43},[33,2821,2822],{"class":113},"1_200",[33,2824,512],{"class":43},[33,2826,2827,2829,2832,2834,2837],{"class":35,"line":884},[33,2828,2800],{"class":43},[33,2830,2831],{"class":113},"RETRIEVED_DOCS",[33,2833,184],{"class":43},[33,2835,2836],{"class":113},"50_000",[33,2838,512],{"class":43},[33,2840,2842,2844,2847,2849,2852],{"class":35,"line":2841},72,[33,2843,2800],{"class":43},[33,2845,2846],{"class":113},"CONVERSATION",[33,2848,184],{"class":43},[33,2850,2851],{"class":113},"30_000",[33,2853,512],{"class":43},[33,2855,2857],{"class":35,"line":2856},73,[33,2858,78],{"emptyLinePlaceholder":77},[33,2860,2862],{"class":35,"line":2861},74,[33,2863,2864],{"class":123},"# Simulate a user pasting a massive document\n",[33,2866,2868,2871,2873],{"class":35,"line":2867},75,[33,2869,2870],{"class":43},"doc_tokens ",[33,2872,193],{"class":39},[33,2874,2875],{"class":113}," 150_000\n",[33,2877,2879,2882,2884,2887,2890],{"class":35,"line":2878},76,[33,2880,2881],{"class":43},"fits ",[33,2883,193],{"class":39},[33,2885,2886],{"class":43}," budget.truncate_to_fit(Section.",[33,2888,2889],{"class":113},"USER_MESSAGE",[33,2891,2892],{"class":43},", doc_tokens)\n",[33,2894,2896],{"class":35,"line":2895},77,[33,2897,2898],{"class":123},"# → only ~118,000 tokens fit — the rest MUST be dropped, summarized, or chunked.\n",[33,2900,2902],{"class":35,"line":2901},78,[33,2903,2904],{"class":123},"# NEVER silently truncate: the caller decides the strategy (summarize? embed+retrieve? error?)\n",[33,2906,2908],{"class":35,"line":2907},79,[33,2909,78],{"emptyLinePlaceholder":77},[33,2911,2913],{"class":35,"line":2912},80,[33,2914,2915],{"class":123},"# --- THE \"lost in the middle\" effect ---\n",[33,2917,2919],{"class":35,"line":2918},81,[33,2920,2921],{"class":123},"# A 200K context window means the API ACCEPTS 200K tokens — NOT that the model\n",[33,2923,2925],{"class":35,"line":2924},82,[33,2926,2927],{"class":123},"# reliably USES all 200K. Attention weight is position-dependent:\n",[33,2929,2931],{"class":35,"line":2930},83,[33,2932,2933],{"class":123},"#   - Beginning tokens: high recall (primacy effect)\n",[33,2935,2937],{"class":35,"line":2936},84,[33,2938,2939],{"class":123},"#   - End tokens: high recall (recency effect)\n",[33,2941,2943],{"class":35,"line":2942},85,[33,2944,2945],{"class":123},"#   - Middle tokens: degraded recall (the \"lost in the middle\" valley)\n",[33,2947,2949],{"class":35,"line":2948},86,[33,2950,2951],{"class":123},"# Production implication: put critical instructions at the START or END of context,\n",[33,2953,2955],{"class":35,"line":2954},87,[33,2956,2957],{"class":123},"# not buried in the middle of a 50-page document dump.\n",[14,2959,2961],{"id":2960},"why-prompting-works-conditioning-the-model","Why Prompting Works: Conditioning the Model",[19,2963,2965],{"filename":2964,"language":22},"conditioning_demo.py",[24,2966,2968],{"className":26,"code":2967,"language":22,"meta":28,"style":28},"\"\"\"\nWhy \"You are an expert tax attorney\" produces better tax answers than \"explain taxes\".\nThe model has no persistent identity — you are selecting a REGION of its learned distribution.\n\"\"\"\nfrom dataclasses import dataclass\n\n@dataclass\nclass PromptAsDistributionSelector:\n    \"\"\"\n    During training, the model saw that text opening like an expert legal explainer\n    is statistically followed by careful, hedged, jargon-appropriate content.\n    Text opening like a casual forum post is followed by casual, less precise content.\n    Your prompt selects WHICH distribution region to sample from.\n    \"\"\"\n    prompt_prefix: str\n    # The prefix conditions P(output | prefix) toward a specific region of learned behavior.\n\n    def expected_output_region(self) -> str:\n        \"\"\"Maps prompt framing to the training-distribution region it activates.\"\"\"\n        regions = {\n            \"You are an expert tax attorney\": \"legal_explainer_region → hedged, precise, cites statutes\",\n            \"explain tax implications\":        \"general_forum_region → casual, may omit edge cases, less hedged\",\n            \"You are a senior Rust engineer\":   \"rust_expert_region → idiomatic code, mentions ownership\u002Flifetimes\",\n            \"write a rust function\":            \"beginner_region → may use .clone() excessively, miss lifetime annotations\",\n        }\n        return regions.get(self.prompt_prefix, \"default_region → average of training distribution\")\n\n# --- The mechanical view ---\n# P(output | \"You are an expert tax attorney. Explain...\")  ≠  P(output | \"explain taxes\")\n#                                                  ↑\n#                          The prefix shifts the conditional probability mass.\n#                          The model doesn't \"become\" a tax attorney —\n#                          it samples from the text distribution that follows\n#                          tax-attorney-style openings in its training data.\n\n# --- Why early tokens are load-bearing ---\n# Generation is autoregressive: token[n] is sampled conditioned on tokens[0..n-1].\n# The FIRST output token disproportionately constrains all subsequent tokens.\n#   If the model starts with \"Based on IRC Section 280A...\" → locked into citation-heavy mode\n#   If the model starts with \"Sure! So basically...\" → locked into casual explainer mode\n# This is why \"the model starts well, it tends to finish well\" — and the inverse.\n\n# --- Anti-pattern: conflicting conditioning ---\n# If your system prompt says \"Be extremely concise\" but your user prompt says\n# \"Explain in exhaustive detail with examples\", the model isn't \"confused\" —\n# it's weighting two REAL, contradictory signals in its context.\n# Whichever has stronger positional\u002Frelevance attention weight wins, unpredictably.\n# Fix: ensure system and user instructions are ALIGNED, not competing.\n",[30,2969,2970,2974,2979,2984,2988,2998,3002,3006,3015,3019,3024,3029,3034,3039,3043,3050,3055,3059,3072,3077,3086,3098,3111,3124,3137,3142,3159,3163,3168,3173,3178,3183,3188,3193,3198,3202,3207,3212,3217,3222,3227,3232,3236,3241,3246,3251,3256,3261],{"__ignoreMap":28},[33,2971,2972],{"class":35,"line":36},[33,2973,1335],{"class":103},[33,2975,2976],{"class":35,"line":47},[33,2977,2978],{"class":103},"Why \"You are an expert tax attorney\" produces better tax answers than \"explain taxes\".\n",[33,2980,2981],{"class":35,"line":61},[33,2982,2983],{"class":103},"The model has no persistent identity — you are selecting a REGION of its learned distribution.\n",[33,2985,2986],{"class":35,"line":74},[33,2987,1335],{"class":103},[33,2989,2990,2992,2994,2996],{"class":35,"line":81},[33,2991,50],{"class":39},[33,2993,66],{"class":43},[33,2995,40],{"class":39},[33,2997,71],{"class":43},[33,2999,3000],{"class":35,"line":88},[33,3001,78],{"emptyLinePlaceholder":77},[33,3003,3004],{"class":35,"line":100},[33,3005,85],{"class":84},[33,3007,3008,3010,3013],{"class":35,"line":107},[33,3009,91],{"class":39},[33,3011,3012],{"class":84}," PromptAsDistributionSelector",[33,3014,97],{"class":43},[33,3016,3017],{"class":35,"line":127},[33,3018,276],{"class":103},[33,3020,3021],{"class":35,"line":144},[33,3022,3023],{"class":103},"    During training, the model saw that text opening like an expert legal explainer\n",[33,3025,3026],{"class":35,"line":159},[33,3027,3028],{"class":103},"    is statistically followed by careful, hedged, jargon-appropriate content.\n",[33,3030,3031],{"class":35,"line":175},[33,3032,3033],{"class":103},"    Text opening like a casual forum post is followed by casual, less precise content.\n",[33,3035,3036],{"class":35,"line":202},[33,3037,3038],{"class":103},"    Your prompt selects WHICH distribution region to sample from.\n",[33,3040,3041],{"class":35,"line":207},[33,3042,276],{"class":103},[33,3044,3045,3048],{"class":35,"line":219},[33,3046,3047],{"class":43},"    prompt_prefix: ",[33,3049,2152],{"class":113},[33,3051,3052],{"class":35,"line":228},[33,3053,3054],{"class":123},"    # The prefix conditions P(output | prefix) toward a specific region of learned behavior.\n",[33,3056,3057],{"class":35,"line":237},[33,3058,78],{"emptyLinePlaceholder":77},[33,3060,3061,3063,3066,3068,3070],{"class":35,"line":243},[33,3062,2209],{"class":39},[33,3064,3065],{"class":84}," expected_output_region",[33,3067,2318],{"class":43},[33,3069,181],{"class":113},[33,3071,97],{"class":43},[33,3073,3074],{"class":35,"line":263},[33,3075,3076],{"class":103},"        \"\"\"Maps prompt framing to the training-distribution region it activates.\"\"\"\n",[33,3078,3079,3082,3084],{"class":35,"line":273},[33,3080,3081],{"class":43},"        regions ",[33,3083,193],{"class":39},[33,3085,1874],{"class":43},[33,3087,3088,3091,3093,3096],{"class":35,"line":279},[33,3089,3090],{"class":103},"            \"You are an expert tax attorney\"",[33,3092,1812],{"class":43},[33,3094,3095],{"class":103},"\"legal_explainer_region → hedged, precise, cites statutes\"",[33,3097,1945],{"class":43},[33,3099,3100,3103,3106,3109],{"class":35,"line":285},[33,3101,3102],{"class":103},"            \"explain tax implications\"",[33,3104,3105],{"class":43},":        ",[33,3107,3108],{"class":103},"\"general_forum_region → casual, may omit edge cases, less hedged\"",[33,3110,1945],{"class":43},[33,3112,3113,3116,3119,3122],{"class":35,"line":291},[33,3114,3115],{"class":103},"            \"You are a senior Rust engineer\"",[33,3117,3118],{"class":43},":   ",[33,3120,3121],{"class":103},"\"rust_expert_region → idiomatic code, mentions ownership\u002Flifetimes\"",[33,3123,1945],{"class":43},[33,3125,3126,3129,3132,3135],{"class":35,"line":296},[33,3127,3128],{"class":103},"            \"write a rust function\"",[33,3130,3131],{"class":43},":            ",[33,3133,3134],{"class":103},"\"beginner_region → may use .clone() excessively, miss lifetime annotations\"",[33,3136,1945],{"class":43},[33,3138,3139],{"class":35,"line":308},[33,3140,3141],{"class":43},"        }\n",[33,3143,3144,3146,3149,3151,3154,3157],{"class":35,"line":322},[33,3145,2332],{"class":39},[33,3147,3148],{"class":43}," regions.get(",[33,3150,2724],{"class":113},[33,3152,3153],{"class":43},".prompt_prefix, ",[33,3155,3156],{"class":103},"\"default_region → average of training distribution\"",[33,3158,512],{"class":43},[33,3160,3161],{"class":35,"line":327},[33,3162,78],{"emptyLinePlaceholder":77},[33,3164,3165],{"class":35,"line":333},[33,3166,3167],{"class":123},"# --- The mechanical view ---\n",[33,3169,3170],{"class":35,"line":339},[33,3171,3172],{"class":123},"# P(output | \"You are an expert tax attorney. Explain...\")  ≠  P(output | \"explain taxes\")\n",[33,3174,3175],{"class":35,"line":345},[33,3176,3177],{"class":123},"#                                                  ↑\n",[33,3179,3180],{"class":35,"line":361},[33,3181,3182],{"class":123},"#                          The prefix shifts the conditional probability mass.\n",[33,3184,3185],{"class":35,"line":380},[33,3186,3187],{"class":123},"#                          The model doesn't \"become\" a tax attorney —\n",[33,3189,3190],{"class":35,"line":385},[33,3191,3192],{"class":123},"#                          it samples from the text distribution that follows\n",[33,3194,3195],{"class":35,"line":391},[33,3196,3197],{"class":123},"#                          tax-attorney-style openings in its training data.\n",[33,3199,3200],{"class":35,"line":406},[33,3201,78],{"emptyLinePlaceholder":77},[33,3203,3204],{"class":35,"line":417},[33,3205,3206],{"class":123},"# --- Why early tokens are load-bearing ---\n",[33,3208,3209],{"class":35,"line":440},[33,3210,3211],{"class":123},"# Generation is autoregressive: token[n] is sampled conditioned on tokens[0..n-1].\n",[33,3213,3214],{"class":35,"line":467},[33,3215,3216],{"class":123},"# The FIRST output token disproportionately constrains all subsequent tokens.\n",[33,3218,3219],{"class":35,"line":472},[33,3220,3221],{"class":123},"#   If the model starts with \"Based on IRC Section 280A...\" → locked into citation-heavy mode\n",[33,3223,3224],{"class":35,"line":478},[33,3225,3226],{"class":123},"#   If the model starts with \"Sure! So basically...\" → locked into casual explainer mode\n",[33,3228,3229],{"class":35,"line":492},[33,3230,3231],{"class":123},"# This is why \"the model starts well, it tends to finish well\" — and the inverse.\n",[33,3233,3234],{"class":35,"line":515},[33,3235,78],{"emptyLinePlaceholder":77},[33,3237,3238],{"class":35,"line":545},[33,3239,3240],{"class":123},"# --- Anti-pattern: conflicting conditioning ---\n",[33,3242,3243],{"class":35,"line":551},[33,3244,3245],{"class":123},"# If your system prompt says \"Be extremely concise\" but your user prompt says\n",[33,3247,3248],{"class":35,"line":581},[33,3249,3250],{"class":123},"# \"Explain in exhaustive detail with examples\", the model isn't \"confused\" —\n",[33,3252,3253],{"class":35,"line":598},[33,3254,3255],{"class":123},"# it's weighting two REAL, contradictory signals in its context.\n",[33,3257,3258],{"class":35,"line":617},[33,3259,3260],{"class":123},"# Whichever has stronger positional\u002Frelevance attention weight wins, unpredictably.\n",[33,3262,3263],{"class":35,"line":631},[33,3264,3265],{"class":123},"# Fix: ensure system and user instructions are ALIGNED, not competing.\n",[14,3267,3269],{"id":3268},"a-production-system-prompt-example","A Production System Prompt Example",[19,3271,3274],{"filename":3272,"language":3273},"triage_system_prompt.md","markdown",[24,3275,3278],{"className":3276,"code":3277,"language":3273,"meta":28,"style":28},"language-markdown shiki shiki-themes github-light github-dark","You are a support-ticket triage assistant for a B2B SaaS company.\n\nYour job: read the incoming support message and classify it into exactly\none of these categories: BILLING, BUG_REPORT, FEATURE_REQUEST, ACCOUNT_ACCESS,\nor OTHER. Then extract the customer's stated urgency (LOW, MEDIUM, HIGH) based\non their own language, not your judgment of how urgent it \"really\" is.\n\nRespond with only a JSON object in this exact shape:\n{\"category\": \"...\", \"urgency\": \"...\", \"summary\": \"one sentence, no more than 20 words\"}\n\nRules:\n- category MUST be one of the five uppercase strings above — no others, ever.\n- urgency MUST be one of: LOW, MEDIUM, HIGH.\n- summary MUST be a single sentence, maximum 20 words.\n- If the message doesn't clearly fit one category, choose the closest one —\n  always pick exactly one, never refuse or say \"unclear\".\n- Do not include any text outside the JSON object. No preamble, no explanation.\n",[30,3279,3280,3285,3289,3294,3299,3304,3309,3313,3318,3323,3327,3332,3339,3346,3353,3360,3365],{"__ignoreMap":28},[33,3281,3282],{"class":35,"line":36},[33,3283,3284],{"class":43},"You are a support-ticket triage assistant for a B2B SaaS company.\n",[33,3286,3287],{"class":35,"line":47},[33,3288,78],{"emptyLinePlaceholder":77},[33,3290,3291],{"class":35,"line":61},[33,3292,3293],{"class":43},"Your job: read the incoming support message and classify it into exactly\n",[33,3295,3296],{"class":35,"line":74},[33,3297,3298],{"class":43},"one of these categories: BILLING, BUG_REPORT, FEATURE_REQUEST, ACCOUNT_ACCESS,\n",[33,3300,3301],{"class":35,"line":81},[33,3302,3303],{"class":43},"or OTHER. Then extract the customer's stated urgency (LOW, MEDIUM, HIGH) based\n",[33,3305,3306],{"class":35,"line":88},[33,3307,3308],{"class":43},"on their own language, not your judgment of how urgent it \"really\" is.\n",[33,3310,3311],{"class":35,"line":100},[33,3312,78],{"emptyLinePlaceholder":77},[33,3314,3315],{"class":35,"line":107},[33,3316,3317],{"class":43},"Respond with only a JSON object in this exact shape:\n",[33,3319,3320],{"class":35,"line":127},[33,3321,3322],{"class":43},"{\"category\": \"...\", \"urgency\": \"...\", \"summary\": \"one sentence, no more than 20 words\"}\n",[33,3324,3325],{"class":35,"line":144},[33,3326,78],{"emptyLinePlaceholder":77},[33,3328,3329],{"class":35,"line":159},[33,3330,3331],{"class":43},"Rules:\n",[33,3333,3334,3336],{"class":35,"line":175},[33,3335,428],{"class":503},[33,3337,3338],{"class":43}," category MUST be one of the five uppercase strings above — no others, ever.\n",[33,3340,3341,3343],{"class":35,"line":202},[33,3342,428],{"class":503},[33,3344,3345],{"class":43}," urgency MUST be one of: LOW, MEDIUM, HIGH.\n",[33,3347,3348,3350],{"class":35,"line":207},[33,3349,428],{"class":503},[33,3351,3352],{"class":43}," summary MUST be a single sentence, maximum 20 words.\n",[33,3354,3355,3357],{"class":35,"line":219},[33,3356,428],{"class":503},[33,3358,3359],{"class":43}," If the message doesn't clearly fit one category, choose the closest one —\n",[33,3361,3362],{"class":35,"line":228},[33,3363,3364],{"class":43},"  always pick exactly one, never refuse or say \"unclear\".\n",[33,3366,3367,3369],{"class":35,"line":237},[33,3368,428],{"class":503},[33,3370,3371],{"class":43}," Do not include any text outside the JSON object. No preamble, no explanation.\n",[19,3373,3375],{"filename":3374,"language":22},"triage_validator.py",[24,3376,3378],{"className":26,"code":3377,"language":22,"meta":28,"style":28},"\"\"\"Client-side validation for the triage system prompt above — never trust raw LLM output.\"\"\"\nimport json\nfrom dataclasses import dataclass\nfrom enum import Enum\n\nclass Category(Enum):\n    BILLING = \"BILLING\"\n    BUG_REPORT = \"BUG_REPORT\"\n    FEATURE_REQUEST = \"FEATURE_REQUEST\"\n    ACCOUNT_ACCESS = \"ACCOUNT_ACCESS\"\n    OTHER = \"OTHER\"\n\nclass Urgency(Enum):\n    LOW = \"LOW\"\n    MEDIUM = \"MEDIUM\"\n    HIGH = \"HIGH\"\n\n@dataclass\nclass TriageResult:\n    category: Category\n    urgency: Urgency\n    summary: str\n\n    @classmethod\n    def from_llm_output(cls, raw: str) -> \"TriageResult\":\n        \"\"\"Parse + validate LLM output. Raises on any deviation from the contract.\"\"\"\n        # Strip any accidental preamble\u002Fepilogue the model might add despite instructions\n        raw = raw.strip()\n        # Find the JSON object even if surrounded by stray text\n        start, end = raw.find(\"{\"), raw.rfind(\"}\")\n        if start == -1 or end == -1:\n            raise ValueError(f\"No JSON object found in LLM output: {raw[:100]!r}\")\n        try:\n            data = json.loads(raw[start : end + 1])\n        except json.JSONDecodeError as e:\n            raise ValueError(f\"Invalid JSON from LLM: {e}\") from e\n\n        # Validate category against the enumerated set — reject anything else\n        cat_str = data.get(\"category\", \"\")\n        try:\n            category = Category(cat_str)\n        except ValueError:\n            raise ValueError(f\"Invalid category {cat_str!r} — must be one of {[c.value for c in Category]}\")\n\n        # Validate urgency\n        urg_str = data.get(\"urgency\", \"\")\n        try:\n            urgency = Urgency(urg_str)\n        except ValueError:\n            raise ValueError(f\"Invalid urgency {urg_str!r} — must be one of {[u.value for u in Urgency]}\")\n\n        summary = data.get(\"summary\", \"\")\n        if not summary or len(summary.split()) > 20:\n            raise ValueError(f\"Summary must be 1-20 words, got {len(summary.split())}: {summary!r}\")\n\n        return cls(category=category, urgency=urgency, summary=summary)\n",[30,3379,3380,3385,3392,3402,3412,3416,3429,3439,3449,3459,3469,3479,3483,3496,3506,3516,3526,3530,3534,3543,3548,3553,3560,3564,3571,3590,3595,3600,3610,3615,3636,3664,3698,3705,3723,3737,3766,3770,3775,3794,3800,3810,3818,3864,3868,3873,3891,3897,3907,3915,3960,3964,3982,4008,4044,4048],{"__ignoreMap":28},[33,3381,3382],{"class":35,"line":36},[33,3383,3384],{"class":103},"\"\"\"Client-side validation for the triage system prompt above — never trust raw LLM output.\"\"\"\n",[33,3386,3387,3389],{"class":35,"line":47},[33,3388,40],{"class":39},[33,3390,3391],{"class":43}," json\n",[33,3393,3394,3396,3398,3400],{"class":35,"line":61},[33,3395,50],{"class":39},[33,3397,66],{"class":43},[33,3399,40],{"class":39},[33,3401,71],{"class":43},[33,3403,3404,3406,3408,3410],{"class":35,"line":74},[33,3405,50],{"class":39},[33,3407,1998],{"class":43},[33,3409,40],{"class":39},[33,3411,2003],{"class":43},[33,3413,3414],{"class":35,"line":81},[33,3415,78],{"emptyLinePlaceholder":77},[33,3417,3418,3420,3423,3425,3427],{"class":35,"line":88},[33,3419,91],{"class":39},[33,3421,3422],{"class":84}," Category",[33,3424,458],{"class":43},[33,3426,2043],{"class":84},[33,3428,2022],{"class":43},[33,3430,3431,3434,3436],{"class":35,"line":100},[33,3432,3433],{"class":113},"    BILLING",[33,3435,117],{"class":39},[33,3437,3438],{"class":103}," \"BILLING\"\n",[33,3440,3441,3444,3446],{"class":35,"line":107},[33,3442,3443],{"class":113},"    BUG_REPORT",[33,3445,117],{"class":39},[33,3447,3448],{"class":103}," \"BUG_REPORT\"\n",[33,3450,3451,3454,3456],{"class":35,"line":127},[33,3452,3453],{"class":113},"    FEATURE_REQUEST",[33,3455,117],{"class":39},[33,3457,3458],{"class":103}," \"FEATURE_REQUEST\"\n",[33,3460,3461,3464,3466],{"class":35,"line":144},[33,3462,3463],{"class":113},"    ACCOUNT_ACCESS",[33,3465,117],{"class":39},[33,3467,3468],{"class":103}," \"ACCOUNT_ACCESS\"\n",[33,3470,3471,3474,3476],{"class":35,"line":159},[33,3472,3473],{"class":113},"    OTHER",[33,3475,117],{"class":39},[33,3477,3478],{"class":103}," \"OTHER\"\n",[33,3480,3481],{"class":35,"line":175},[33,3482,78],{"emptyLinePlaceholder":77},[33,3484,3485,3487,3490,3492,3494],{"class":35,"line":202},[33,3486,91],{"class":39},[33,3488,3489],{"class":84}," Urgency",[33,3491,458],{"class":43},[33,3493,2043],{"class":84},[33,3495,2022],{"class":43},[33,3497,3498,3501,3503],{"class":35,"line":207},[33,3499,3500],{"class":113},"    LOW",[33,3502,117],{"class":39},[33,3504,3505],{"class":103}," \"LOW\"\n",[33,3507,3508,3511,3513],{"class":35,"line":219},[33,3509,3510],{"class":113},"    MEDIUM",[33,3512,117],{"class":39},[33,3514,3515],{"class":103}," \"MEDIUM\"\n",[33,3517,3518,3521,3523],{"class":35,"line":228},[33,3519,3520],{"class":113},"    HIGH",[33,3522,117],{"class":39},[33,3524,3525],{"class":103}," \"HIGH\"\n",[33,3527,3528],{"class":35,"line":237},[33,3529,78],{"emptyLinePlaceholder":77},[33,3531,3532],{"class":35,"line":243},[33,3533,85],{"class":84},[33,3535,3536,3538,3541],{"class":35,"line":263},[33,3537,91],{"class":39},[33,3539,3540],{"class":84}," TriageResult",[33,3542,97],{"class":43},[33,3544,3545],{"class":35,"line":273},[33,3546,3547],{"class":43},"    category: Category\n",[33,3549,3550],{"class":35,"line":279},[33,3551,3552],{"class":43},"    urgency: Urgency\n",[33,3554,3555,3558],{"class":35,"line":285},[33,3556,3557],{"class":43},"    summary: ",[33,3559,2152],{"class":113},[33,3561,3562],{"class":35,"line":291},[33,3563,78],{"emptyLinePlaceholder":77},[33,3565,3566,3568],{"class":35,"line":296},[33,3567,2305],{"class":84},[33,3569,3570],{"class":113},"classmethod\n",[33,3572,3573,3575,3578,3581,3583,3585,3588],{"class":35,"line":308},[33,3574,2209],{"class":39},[33,3576,3577],{"class":84}," from_llm_output",[33,3579,3580],{"class":43},"(cls, raw: ",[33,3582,181],{"class":113},[33,3584,266],{"class":43},[33,3586,3587],{"class":103},"\"TriageResult\"",[33,3589,97],{"class":43},[33,3591,3592],{"class":35,"line":322},[33,3593,3594],{"class":103},"        \"\"\"Parse + validate LLM output. Raises on any deviation from the contract.\"\"\"\n",[33,3596,3597],{"class":35,"line":327},[33,3598,3599],{"class":123},"        # Strip any accidental preamble\u002Fepilogue the model might add despite instructions\n",[33,3601,3602,3605,3607],{"class":35,"line":333},[33,3603,3604],{"class":43},"        raw ",[33,3606,193],{"class":39},[33,3608,3609],{"class":43}," raw.strip()\n",[33,3611,3612],{"class":35,"line":339},[33,3613,3614],{"class":123},"        # Find the JSON object even if surrounded by stray text\n",[33,3616,3617,3620,3622,3625,3628,3631,3634],{"class":35,"line":345},[33,3618,3619],{"class":43},"        start, end ",[33,3621,193],{"class":39},[33,3623,3624],{"class":43}," raw.find(",[33,3626,3627],{"class":103},"\"{\"",[33,3629,3630],{"class":43},"), raw.rfind(",[33,3632,3633],{"class":103},"\"}\"",[33,3635,512],{"class":43},[33,3637,3638,3640,3643,3645,3648,3650,3653,3656,3658,3660,3662],{"class":35,"line":361},[33,3639,829],{"class":39},[33,3641,3642],{"class":43}," start ",[33,3644,819],{"class":39},[33,3646,3647],{"class":39}," -",[33,3649,431],{"class":113},[33,3651,3652],{"class":39}," or",[33,3654,3655],{"class":43}," end ",[33,3657,819],{"class":39},[33,3659,3647],{"class":39},[33,3661,431],{"class":113},[33,3663,97],{"class":43},[33,3665,3666,3668,3671,3673,3675,3678,3680,3683,3686,3689,3692,3694,3696],{"class":35,"line":380},[33,3667,2238],{"class":39},[33,3669,3670],{"class":113}," ValueError",[33,3672,458],{"class":43},[33,3674,2503],{"class":39},[33,3676,3677],{"class":103},"\"No JSON object found in LLM output: ",[33,3679,2509],{"class":113},[33,3681,3682],{"class":43},"raw[:",[33,3684,3685],{"class":113},"100",[33,3687,3688],{"class":43},"]",[33,3690,3691],{"class":39},"!r",[33,3693,2258],{"class":113},[33,3695,2517],{"class":103},[33,3697,512],{"class":43},[33,3699,3700,3703],{"class":35,"line":385},[33,3701,3702],{"class":39},"        try",[33,3704,97],{"class":43},[33,3706,3707,3710,3712,3715,3717,3720],{"class":35,"line":391},[33,3708,3709],{"class":43},"            data ",[33,3711,193],{"class":39},[33,3713,3714],{"class":43}," json.loads(raw[start : end ",[33,3716,2534],{"class":39},[33,3718,3719],{"class":113}," 1",[33,3721,3722],{"class":43},"])\n",[33,3724,3725,3728,3731,3734],{"class":35,"line":406},[33,3726,3727],{"class":39},"        except",[33,3729,3730],{"class":43}," json.JSONDecodeError ",[33,3732,3733],{"class":39},"as",[33,3735,3736],{"class":43}," e:\n",[33,3738,3739,3741,3743,3745,3747,3750,3752,3755,3757,3759,3761,3763],{"class":35,"line":417},[33,3740,2238],{"class":39},[33,3742,3670],{"class":113},[33,3744,458],{"class":43},[33,3746,2503],{"class":39},[33,3748,3749],{"class":103},"\"Invalid JSON from LLM: ",[33,3751,2509],{"class":113},[33,3753,3754],{"class":43},"e",[33,3756,2258],{"class":113},[33,3758,2517],{"class":103},[33,3760,573],{"class":43},[33,3762,50],{"class":39},[33,3764,3765],{"class":43}," e\n",[33,3767,3768],{"class":35,"line":440},[33,3769,78],{"emptyLinePlaceholder":77},[33,3771,3772],{"class":35,"line":467},[33,3773,3774],{"class":123},"        # Validate category against the enumerated set — reject anything else\n",[33,3776,3777,3780,3782,3785,3788,3790,3792],{"class":35,"line":472},[33,3778,3779],{"class":43},"        cat_str ",[33,3781,193],{"class":39},[33,3783,3784],{"class":43}," data.get(",[33,3786,3787],{"class":103},"\"category\"",[33,3789,184],{"class":43},[33,3791,1657],{"class":103},[33,3793,512],{"class":43},[33,3795,3796,3798],{"class":35,"line":478},[33,3797,3702],{"class":39},[33,3799,97],{"class":43},[33,3801,3802,3805,3807],{"class":35,"line":492},[33,3803,3804],{"class":43},"            category ",[33,3806,193],{"class":39},[33,3808,3809],{"class":43}," Category(cat_str)\n",[33,3811,3812,3814,3816],{"class":35,"line":515},[33,3813,3727],{"class":39},[33,3815,3670],{"class":113},[33,3817,97],{"class":43},[33,3819,3820,3822,3824,3826,3828,3831,3833,3836,3838,3840,3843,3845,3848,3850,3853,3855,3858,3860,3862],{"class":35,"line":545},[33,3821,2238],{"class":39},[33,3823,3670],{"class":113},[33,3825,458],{"class":43},[33,3827,2503],{"class":39},[33,3829,3830],{"class":103},"\"Invalid category ",[33,3832,2509],{"class":113},[33,3834,3835],{"class":43},"cat_str",[33,3837,3691],{"class":39},[33,3839,2258],{"class":113},[33,3841,3842],{"class":103}," — must be one of ",[33,3844,2509],{"class":113},[33,3846,3847],{"class":43},"[c.value ",[33,3849,2379],{"class":39},[33,3851,3852],{"class":43}," c ",[33,3854,791],{"class":39},[33,3856,3857],{"class":43}," Category]",[33,3859,2258],{"class":113},[33,3861,2517],{"class":103},[33,3863,512],{"class":43},[33,3865,3866],{"class":35,"line":551},[33,3867,78],{"emptyLinePlaceholder":77},[33,3869,3870],{"class":35,"line":581},[33,3871,3872],{"class":123},"        # Validate urgency\n",[33,3874,3875,3878,3880,3882,3885,3887,3889],{"class":35,"line":598},[33,3876,3877],{"class":43},"        urg_str ",[33,3879,193],{"class":39},[33,3881,3784],{"class":43},[33,3883,3884],{"class":103},"\"urgency\"",[33,3886,184],{"class":43},[33,3888,1657],{"class":103},[33,3890,512],{"class":43},[33,3892,3893,3895],{"class":35,"line":617},[33,3894,3702],{"class":39},[33,3896,97],{"class":43},[33,3898,3899,3902,3904],{"class":35,"line":631},[33,3900,3901],{"class":43},"            urgency ",[33,3903,193],{"class":39},[33,3905,3906],{"class":43}," Urgency(urg_str)\n",[33,3908,3909,3911,3913],{"class":35,"line":636},[33,3910,3727],{"class":39},[33,3912,3670],{"class":113},[33,3914,97],{"class":43},[33,3916,3917,3919,3921,3923,3925,3928,3930,3933,3935,3937,3939,3941,3944,3946,3949,3951,3954,3956,3958],{"class":35,"line":642},[33,3918,2238],{"class":39},[33,3920,3670],{"class":113},[33,3922,458],{"class":43},[33,3924,2503],{"class":39},[33,3926,3927],{"class":103},"\"Invalid urgency ",[33,3929,2509],{"class":113},[33,3931,3932],{"class":43},"urg_str",[33,3934,3691],{"class":39},[33,3936,2258],{"class":113},[33,3938,3842],{"class":103},[33,3940,2509],{"class":113},[33,3942,3943],{"class":43},"[u.value ",[33,3945,2379],{"class":39},[33,3947,3948],{"class":43}," u ",[33,3950,791],{"class":39},[33,3952,3953],{"class":43}," Urgency]",[33,3955,2258],{"class":113},[33,3957,2517],{"class":103},[33,3959,512],{"class":43},[33,3961,3962],{"class":35,"line":665},[33,3963,78],{"emptyLinePlaceholder":77},[33,3965,3966,3969,3971,3973,3976,3978,3980],{"class":35,"line":686},[33,3967,3968],{"class":43},"        summary ",[33,3970,193],{"class":39},[33,3972,3784],{"class":43},[33,3974,3975],{"class":103},"\"summary\"",[33,3977,184],{"class":43},[33,3979,1657],{"class":103},[33,3981,512],{"class":43},[33,3983,3984,3986,3989,3992,3995,3998,4001,4003,4006],{"class":35,"line":691},[33,3985,829],{"class":39},[33,3987,3988],{"class":39}," not",[33,3990,3991],{"class":43}," summary ",[33,3993,3994],{"class":39},"or",[33,3996,3997],{"class":113}," len",[33,3999,4000],{"class":43},"(summary.split()) ",[33,4002,399],{"class":39},[33,4004,4005],{"class":113}," 20",[33,4007,97],{"class":43},[33,4009,4010,4012,4014,4016,4018,4021,4024,4027,4029,4031,4033,4036,4038,4040,4042],{"class":35,"line":703},[33,4011,2238],{"class":39},[33,4013,3670],{"class":113},[33,4015,458],{"class":43},[33,4017,2503],{"class":39},[33,4019,4020],{"class":103},"\"Summary must be 1-20 words, got ",[33,4022,4023],{"class":113},"{len",[33,4025,4026],{"class":43},"(summary.split())",[33,4028,2258],{"class":113},[33,4030,1812],{"class":103},[33,4032,2509],{"class":113},[33,4034,4035],{"class":43},"summary",[33,4037,3691],{"class":39},[33,4039,2258],{"class":113},[33,4041,2517],{"class":103},[33,4043,512],{"class":43},[33,4045,4046],{"class":35,"line":709},[33,4047,78],{"emptyLinePlaceholder":77},[33,4049,4050,4052,4055,4057,4060,4062,4065,4068,4070,4073,4075,4077],{"class":35,"line":715},[33,4051,2332],{"class":39},[33,4053,4054],{"class":113}," cls",[33,4056,458],{"class":43},[33,4058,4059],{"class":503},"category",[33,4061,193],{"class":39},[33,4063,4064],{"class":43},"category, ",[33,4066,4067],{"class":503},"urgency",[33,4069,193],{"class":39},[33,4071,4072],{"class":43},"urgency, ",[33,4074,4035],{"class":503},[33,4076,193],{"class":39},[33,4078,4079],{"class":43},"summary)\n",[14,4081,4083],{"id":4082},"tips-tricks","💡 Tips & Tricks",[19,4085,4087],{"filename":4086,"language":22},"tips_and_tricks.py",[24,4088,4090],{"className":26,"code":4089,"language":22,"meta":28,"style":28},"# ─── [Performance] Token budgets are asymmetric: input is cheaper than output ───\n# Most providers charge 3-5x more for output tokens than input tokens.\n# Strategy: invest in MORE input context (examples, constraints, retrieved docs)\n# to get SHORTER, more targeted output — cheaper AND higher quality.\n#   BAD:  sparse prompt → model writes 800 output tokens exploring → expensive + verbose\n#   GOOD: rich prompt with 5 examples → model writes 50 output tokens matching pattern → cheap + precise\n\n# ─── [Debug] The \"what text follows this?\" mental model ───\n# When output is bad, don't ask \"why doesn't it understand me?\"\n# Ask: \"In the training distribution, what text statistically follows what I wrote?\"\n#   You wrote an open-ended question → training data says open-ended questions get long rambling answers\n#   Fix: constrain the output format so the continuation space is narrow\n\n# ─── [Idiom] Use the provider's real tokenizer, not a heuristic ───\n# import tiktoken\n# enc = tiktoken.encoding_for_model(\"gpt-4o\")\n# exact_count = len(enc.encode(your_text))\n# Never use len(text.split()) or len(text) \u002F\u002F 4 for billing-critical calculations.\n# Differences compound across long documents and across providers.\n\n# ─── [Performance] Early tokens are load-bearing — front-load constraints ───\n# Because generation is left-to-right autoregressive, the first sentence of the\n# model's output shapes everything after it. If you can influence the opening\n# (via formatting instructions or a strong constraint on the first line), do it.\n# \"Start your response with the JSON object. No preamble.\" ← this is a performance optimization.\n\n# ─── [Safety] \"Temperature 0\" is NOT determinism ───\n# Even greedy decoding (argmax) can produce different outputs across runs due to:\n#   - floating-point non-determinism in GPU kernels\n#   - batch-dependent parallelism (same request in different batches → different rounding)\n#   - provider-side model versioning (silent weight updates between calls)\n# Never build a system that assumes bit-for-bit reproducibility. Always have a\n# validation layer that checks output STRUCTURE, not exact string equality.\n\n# ─── [Idiom] Append boilerplate in post-processing, not in the prompt ───\n# If every response needs a fixed footer (survey link, disclaimer), DON'T ask\n# the model to generate it — append it in your application code after the API call.\n# Why: forcing the model to emit fixed text BEFORE its substantive answer conditions\n# every subsequent token on irrelevant context, degrading answer quality.\n",[30,4091,4092,4097,4102,4107,4112,4117,4122,4126,4131,4136,4141,4146,4151,4155,4160,4165,4170,4175,4183,4188,4192,4197,4202,4207,4212,4217,4221,4226,4231,4236,4241,4246,4251,4256,4260,4265,4270,4275,4280],{"__ignoreMap":28},[33,4093,4094],{"class":35,"line":36},[33,4095,4096],{"class":123},"# ─── [Performance] Token budgets are asymmetric: input is cheaper than output ───\n",[33,4098,4099],{"class":35,"line":47},[33,4100,4101],{"class":123},"# Most providers charge 3-5x more for output tokens than input tokens.\n",[33,4103,4104],{"class":35,"line":61},[33,4105,4106],{"class":123},"# Strategy: invest in MORE input context (examples, constraints, retrieved docs)\n",[33,4108,4109],{"class":35,"line":74},[33,4110,4111],{"class":123},"# to get SHORTER, more targeted output — cheaper AND higher quality.\n",[33,4113,4114],{"class":35,"line":81},[33,4115,4116],{"class":123},"#   BAD:  sparse prompt → model writes 800 output tokens exploring → expensive + verbose\n",[33,4118,4119],{"class":35,"line":88},[33,4120,4121],{"class":123},"#   GOOD: rich prompt with 5 examples → model writes 50 output tokens matching pattern → cheap + precise\n",[33,4123,4124],{"class":35,"line":100},[33,4125,78],{"emptyLinePlaceholder":77},[33,4127,4128],{"class":35,"line":107},[33,4129,4130],{"class":123},"# ─── [Debug] The \"what text follows this?\" mental model ───\n",[33,4132,4133],{"class":35,"line":127},[33,4134,4135],{"class":123},"# When output is bad, don't ask \"why doesn't it understand me?\"\n",[33,4137,4138],{"class":35,"line":144},[33,4139,4140],{"class":123},"# Ask: \"In the training distribution, what text statistically follows what I wrote?\"\n",[33,4142,4143],{"class":35,"line":159},[33,4144,4145],{"class":123},"#   You wrote an open-ended question → training data says open-ended questions get long rambling answers\n",[33,4147,4148],{"class":35,"line":175},[33,4149,4150],{"class":123},"#   Fix: constrain the output format so the continuation space is narrow\n",[33,4152,4153],{"class":35,"line":202},[33,4154,78],{"emptyLinePlaceholder":77},[33,4156,4157],{"class":35,"line":207},[33,4158,4159],{"class":123},"# ─── [Idiom] Use the provider's real tokenizer, not a heuristic ───\n",[33,4161,4162],{"class":35,"line":219},[33,4163,4164],{"class":123},"# import tiktoken\n",[33,4166,4167],{"class":35,"line":228},[33,4168,4169],{"class":123},"# enc = tiktoken.encoding_for_model(\"gpt-4o\")\n",[33,4171,4172],{"class":35,"line":237},[33,4173,4174],{"class":123},"# exact_count = len(enc.encode(your_text))\n",[33,4176,4177,4180],{"class":35,"line":243},[33,4178,4179],{"class":123},"# Never use len(text.split()) or len(text)",[33,4181,4182],{"class":123}," \u002F\u002F 4 for billing-critical calculations.\n",[33,4184,4185],{"class":35,"line":263},[33,4186,4187],{"class":123},"# Differences compound across long documents and across providers.\n",[33,4189,4190],{"class":35,"line":273},[33,4191,78],{"emptyLinePlaceholder":77},[33,4193,4194],{"class":35,"line":279},[33,4195,4196],{"class":123},"# ─── [Performance] Early tokens are load-bearing — front-load constraints ───\n",[33,4198,4199],{"class":35,"line":285},[33,4200,4201],{"class":123},"# Because generation is left-to-right autoregressive, the first sentence of the\n",[33,4203,4204],{"class":35,"line":291},[33,4205,4206],{"class":123},"# model's output shapes everything after it. If you can influence the opening\n",[33,4208,4209],{"class":35,"line":296},[33,4210,4211],{"class":123},"# (via formatting instructions or a strong constraint on the first line), do it.\n",[33,4213,4214],{"class":35,"line":308},[33,4215,4216],{"class":123},"# \"Start your response with the JSON object. No preamble.\" ← this is a performance optimization.\n",[33,4218,4219],{"class":35,"line":322},[33,4220,78],{"emptyLinePlaceholder":77},[33,4222,4223],{"class":35,"line":327},[33,4224,4225],{"class":123},"# ─── [Safety] \"Temperature 0\" is NOT determinism ───\n",[33,4227,4228],{"class":35,"line":333},[33,4229,4230],{"class":123},"# Even greedy decoding (argmax) can produce different outputs across runs due to:\n",[33,4232,4233],{"class":35,"line":339},[33,4234,4235],{"class":123},"#   - floating-point non-determinism in GPU kernels\n",[33,4237,4238],{"class":35,"line":345},[33,4239,4240],{"class":123},"#   - batch-dependent parallelism (same request in different batches → different rounding)\n",[33,4242,4243],{"class":35,"line":361},[33,4244,4245],{"class":123},"#   - provider-side model versioning (silent weight updates between calls)\n",[33,4247,4248],{"class":35,"line":380},[33,4249,4250],{"class":123},"# Never build a system that assumes bit-for-bit reproducibility. Always have a\n",[33,4252,4253],{"class":35,"line":385},[33,4254,4255],{"class":123},"# validation layer that checks output STRUCTURE, not exact string equality.\n",[33,4257,4258],{"class":35,"line":391},[33,4259,78],{"emptyLinePlaceholder":77},[33,4261,4262],{"class":35,"line":406},[33,4263,4264],{"class":123},"# ─── [Idiom] Append boilerplate in post-processing, not in the prompt ───\n",[33,4266,4267],{"class":35,"line":417},[33,4268,4269],{"class":123},"# If every response needs a fixed footer (survey link, disclaimer), DON'T ask\n",[33,4271,4272],{"class":35,"line":440},[33,4273,4274],{"class":123},"# the model to generate it — append it in your application code after the API call.\n",[33,4276,4277],{"class":35,"line":467},[33,4278,4279],{"class":123},"# Why: forcing the model to emit fixed text BEFORE its substantive answer conditions\n",[33,4281,4282],{"class":35,"line":472},[33,4283,4284],{"class":123},"# every subsequent token on irrelevant context, degrading answer quality.\n",[14,4286,4288],{"id":4287},"️-edge-cases-gotchas","⚠️ Edge Cases & Gotchas",[19,4290,4292],{"filename":4291,"language":22},"edge_cases.py",[24,4293,4295],{"className":26,"code":4294,"language":22,"meta":28,"style":28},"# ─── [Gotcha] Empty \u002F whitespace-only prompts produce chaos ───\ndef validate_user_input(prompt: str) -> str:\n    \"\"\"Never send empty input to the model — it has no conditioning signal.\"\"\"\n    if not prompt or not prompt.strip():\n        raise ValueError(\"Empty prompt — model has zero conditioning, output is unpredictable.\")\n    return prompt.strip()\n# Without validation: empty string → model outputs \"How can I help you?\" or hallucinated content\n# With validation: fail explicitly, let the caller decide (default message? retry? error to user?)\n\n# ─── [Gotcha] Silent truncation of the PROMPT itself, not just the answer ───\n# If your input prompt approaches the context limit, naive client libraries may\n# truncate the PROMPT — potentially cutting off your instructions before the task.\n# The model then answers a question it never fully received. Output quality craters\n# with no error message. Always check limits and fail loudly.\ndef safe_prompt_send(prompt: str, token_count_fn, max_input_tokens: int) -> str:\n    count = token_count_fn(prompt)\n    if count > max_input_tokens:\n        raise ValueError(\n            f\"Prompt is {count} tokens, exceeds input limit {max_input_tokens}. \"\n            f\"Truncating would cut instructions — refusing to send.\"\n        )\n    return prompt  # safe to send\n\n# ─── [Gotcha] \"Reasoning looks right, arithmetic is wrong\" ───\n# A model can write a flawless proof and then botch 54321 * 98765.\n# Multi-digit arithmetic is token-pattern completion, not digit-by-digit computation.\n# The model might tokenize \"54321\" as [\"543\", \"21\"] — it never \"sees\" individual digits.\n# Fix: NEVER trust unaided LLM arithmetic for anything that matters.\n#       Use a tool call \u002F code execution for any numeric computation.\n#       result = llm_with_tools.generate(\"Calculate 54321 * 98765\", tools=[calculator_tool])\n\n# ─── [Safety] Non-English text costs 2-3x more tokens for the same content ───\n# A per-message token cap tuned for English users will truncate Japanese\u002FKorean\u002FArabic\u002FHindi\n# users far more aggressively — their messages hit the cap at half the semantic content.\n# Fix: use character-aware or language-aware limits, not a flat token cap.\ndef adaptive_token_limit(text: str, base_limit: int) -> int:\n    \"\"\"Expand the token budget for non-Latin scripts that tokenize inefficiently.\"\"\"\n    non_latin_ratio = sum(1 for c in text if ord(c) > 0x2E80) \u002F max(len(text), 1)\n    # 0x2E80 = start of CJK radicals; rough proxy for non-Latin scripts\n    if non_latin_ratio > 0.3:\n        return int(base_limit * 2.5)  # non-Latin text needs more tokens for same meaning\n    return base_limit\n\n# ─── [Gotcha] Context window ≠ effective recall window ───\n# A 200K-token context window means the API ACCEPTS 200K tokens — that's all the\n# guarantee gives you. It does NOT mean the model reliably RETRIEVES a fact from\n# token position 100,000 in a 200K-token input.\n# The \"lost in the middle\" effect: recall degrades for mid-context positions.\n# Production fix: place critical info at the START (primacy) or END (recency) of context.\n#   system_prompt → [critical constraints here, position 0-800 tokens]\n#   retrieved_docs → [bulk context, lower recall expected]\n#   user_message → [the actual task, near the end, high recency recall]\n",[30,4296,4297,4302,4320,4325,4341,4355,4362,4367,4372,4376,4381,4386,4391,4396,4401,4423,4433,4445,4453,4480,4487,4492,4502,4506,4511,4516,4521,4526,4531,4536,4541,4545,4550,4555,4560,4565,4587,4592,4648,4653,4667,4689,4696,4700,4705,4710,4715,4720,4725,4730,4735,4740],{"__ignoreMap":28},[33,4298,4299],{"class":35,"line":36},[33,4300,4301],{"class":123},"# ─── [Gotcha] Empty \u002F whitespace-only prompts produce chaos ───\n",[33,4303,4304,4306,4309,4312,4314,4316,4318],{"class":35,"line":47},[33,4305,210],{"class":39},[33,4307,4308],{"class":84}," validate_user_input",[33,4310,4311],{"class":43},"(prompt: ",[33,4313,181],{"class":113},[33,4315,266],{"class":43},[33,4317,181],{"class":113},[33,4319,97],{"class":43},[33,4321,4322],{"class":35,"line":61},[33,4323,4324],{"class":103},"    \"\"\"Never send empty input to the model — it has no conditioning signal.\"\"\"\n",[33,4326,4327,4329,4331,4334,4336,4338],{"class":35,"line":74},[33,4328,348],{"class":39},[33,4330,3988],{"class":39},[33,4332,4333],{"class":43}," prompt ",[33,4335,3994],{"class":39},[33,4337,3988],{"class":39},[33,4339,4340],{"class":43}," prompt.strip():\n",[33,4342,4343,4346,4348,4350,4353],{"class":35,"line":81},[33,4344,4345],{"class":39},"        raise",[33,4347,3670],{"class":113},[33,4349,458],{"class":43},[33,4351,4352],{"class":103},"\"Empty prompt — model has zero conditioning, output is unpredictable.\"",[33,4354,512],{"class":43},[33,4356,4357,4359],{"class":35,"line":88},[33,4358,718],{"class":39},[33,4360,4361],{"class":43}," prompt.strip()\n",[33,4363,4364],{"class":35,"line":100},[33,4365,4366],{"class":123},"# Without validation: empty string → model outputs \"How can I help you?\" or hallucinated content\n",[33,4368,4369],{"class":35,"line":107},[33,4370,4371],{"class":123},"# With validation: fail explicitly, let the caller decide (default message? retry? error to user?)\n",[33,4373,4374],{"class":35,"line":127},[33,4375,78],{"emptyLinePlaceholder":77},[33,4377,4378],{"class":35,"line":144},[33,4379,4380],{"class":123},"# ─── [Gotcha] Silent truncation of the PROMPT itself, not just the answer ───\n",[33,4382,4383],{"class":35,"line":159},[33,4384,4385],{"class":123},"# If your input prompt approaches the context limit, naive client libraries may\n",[33,4387,4388],{"class":35,"line":175},[33,4389,4390],{"class":123},"# truncate the PROMPT — potentially cutting off your instructions before the task.\n",[33,4392,4393],{"class":35,"line":202},[33,4394,4395],{"class":123},"# The model then answers a question it never fully received. Output quality craters\n",[33,4397,4398],{"class":35,"line":207},[33,4399,4400],{"class":123},"# with no error message. Always check limits and fail loudly.\n",[33,4402,4403,4405,4408,4410,4412,4415,4417,4419,4421],{"class":35,"line":219},[33,4404,210],{"class":39},[33,4406,4407],{"class":84}," safe_prompt_send",[33,4409,4311],{"class":43},[33,4411,181],{"class":113},[33,4413,4414],{"class":43},", token_count_fn, max_input_tokens: ",[33,4416,133],{"class":113},[33,4418,266],{"class":43},[33,4420,181],{"class":113},[33,4422,97],{"class":43},[33,4424,4425,4428,4430],{"class":35,"line":228},[33,4426,4427],{"class":43},"    count ",[33,4429,193],{"class":39},[33,4431,4432],{"class":43}," token_count_fn(prompt)\n",[33,4434,4435,4437,4440,4442],{"class":35,"line":237},[33,4436,348],{"class":39},[33,4438,4439],{"class":43}," count ",[33,4441,399],{"class":39},[33,4443,4444],{"class":43}," max_input_tokens:\n",[33,4446,4447,4449,4451],{"class":35,"line":243},[33,4448,4345],{"class":39},[33,4450,3670],{"class":113},[33,4452,216],{"class":43},[33,4454,4455,4458,4461,4463,4466,4468,4471,4473,4476,4478],{"class":35,"line":263},[33,4456,4457],{"class":39},"            f",[33,4459,4460],{"class":103},"\"Prompt is ",[33,4462,2509],{"class":113},[33,4464,4465],{"class":43},"count",[33,4467,2258],{"class":113},[33,4469,4470],{"class":103}," tokens, exceeds input limit ",[33,4472,2509],{"class":113},[33,4474,4475],{"class":43},"max_input_tokens",[33,4477,2258],{"class":113},[33,4479,2643],{"class":103},[33,4481,4482,4484],{"class":35,"line":273},[33,4483,4457],{"class":39},[33,4485,4486],{"class":103},"\"Truncating would cut instructions — refusing to send.\"\n",[33,4488,4489],{"class":35,"line":279},[33,4490,4491],{"class":43},"        )\n",[33,4493,4494,4496,4499],{"class":35,"line":285},[33,4495,718],{"class":39},[33,4497,4498],{"class":43}," prompt  ",[33,4500,4501],{"class":123},"# safe to send\n",[33,4503,4504],{"class":35,"line":291},[33,4505,78],{"emptyLinePlaceholder":77},[33,4507,4508],{"class":35,"line":296},[33,4509,4510],{"class":123},"# ─── [Gotcha] \"Reasoning looks right, arithmetic is wrong\" ───\n",[33,4512,4513],{"class":35,"line":308},[33,4514,4515],{"class":123},"# A model can write a flawless proof and then botch 54321 * 98765.\n",[33,4517,4518],{"class":35,"line":322},[33,4519,4520],{"class":123},"# Multi-digit arithmetic is token-pattern completion, not digit-by-digit computation.\n",[33,4522,4523],{"class":35,"line":327},[33,4524,4525],{"class":123},"# The model might tokenize \"54321\" as [\"543\", \"21\"] — it never \"sees\" individual digits.\n",[33,4527,4528],{"class":35,"line":333},[33,4529,4530],{"class":123},"# Fix: NEVER trust unaided LLM arithmetic for anything that matters.\n",[33,4532,4533],{"class":35,"line":339},[33,4534,4535],{"class":123},"#       Use a tool call \u002F code execution for any numeric computation.\n",[33,4537,4538],{"class":35,"line":345},[33,4539,4540],{"class":123},"#       result = llm_with_tools.generate(\"Calculate 54321 * 98765\", tools=[calculator_tool])\n",[33,4542,4543],{"class":35,"line":361},[33,4544,78],{"emptyLinePlaceholder":77},[33,4546,4547],{"class":35,"line":380},[33,4548,4549],{"class":123},"# ─── [Safety] Non-English text costs 2-3x more tokens for the same content ───\n",[33,4551,4552],{"class":35,"line":385},[33,4553,4554],{"class":123},"# A per-message token cap tuned for English users will truncate Japanese\u002FKorean\u002FArabic\u002FHindi\n",[33,4556,4557],{"class":35,"line":391},[33,4558,4559],{"class":123},"# users far more aggressively — their messages hit the cap at half the semantic content.\n",[33,4561,4562],{"class":35,"line":406},[33,4563,4564],{"class":123},"# Fix: use character-aware or language-aware limits, not a flat token cap.\n",[33,4566,4567,4569,4572,4574,4576,4579,4581,4583,4585],{"class":35,"line":417},[33,4568,210],{"class":39},[33,4570,4571],{"class":84}," adaptive_token_limit",[33,4573,1418],{"class":43},[33,4575,181],{"class":113},[33,4577,4578],{"class":43},", base_limit: ",[33,4580,133],{"class":113},[33,4582,266],{"class":43},[33,4584,133],{"class":113},[33,4586,97],{"class":43},[33,4588,4589],{"class":35,"line":440},[33,4590,4591],{"class":103},"    \"\"\"Expand the token budget for non-Latin scripts that tokenize inefficiently.\"\"\"\n",[33,4593,4594,4597,4599,4601,4603,4605,4608,4610,4612,4615,4617,4620,4623,4625,4628,4631,4633,4635,4637,4639,4641,4644,4646],{"class":35,"line":467},[33,4595,4596],{"class":43},"    non_latin_ratio ",[33,4598,193],{"class":39},[33,4600,2373],{"class":113},[33,4602,458],{"class":43},[33,4604,431],{"class":113},[33,4606,4607],{"class":39}," for",[33,4609,3852],{"class":43},[33,4611,791],{"class":39},[33,4613,4614],{"class":43}," text ",[33,4616,2392],{"class":39},[33,4618,4619],{"class":113}," ord",[33,4621,4622],{"class":43},"(c) ",[33,4624,399],{"class":39},[33,4626,4627],{"class":39}," 0x",[33,4629,4630],{"class":113},"2E80",[33,4632,573],{"class":43},[33,4634,371],{"class":39},[33,4636,1849],{"class":113},[33,4638,458],{"class":43},[33,4640,1858],{"class":113},[33,4642,4643],{"class":43},"(text), ",[33,4645,431],{"class":113},[33,4647,512],{"class":43},[33,4649,4650],{"class":35,"line":472},[33,4651,4652],{"class":123},"    # 0x2E80 = start of CJK radicals; rough proxy for non-Latin scripts\n",[33,4654,4655,4657,4660,4662,4665],{"class":35,"line":478},[33,4656,348],{"class":39},[33,4658,4659],{"class":43}," non_latin_ratio ",[33,4661,399],{"class":39},[33,4663,4664],{"class":113}," 0.3",[33,4666,97],{"class":43},[33,4668,4669,4671,4674,4677,4680,4683,4686],{"class":35,"line":492},[33,4670,2332],{"class":39},[33,4672,4673],{"class":113}," int",[33,4675,4676],{"class":43},"(base_limit ",[33,4678,4679],{"class":39},"*",[33,4681,4682],{"class":113}," 2.5",[33,4684,4685],{"class":43},")  ",[33,4687,4688],{"class":123},"# non-Latin text needs more tokens for same meaning\n",[33,4690,4691,4693],{"class":35,"line":515},[33,4692,718],{"class":39},[33,4694,4695],{"class":43}," base_limit\n",[33,4697,4698],{"class":35,"line":545},[33,4699,78],{"emptyLinePlaceholder":77},[33,4701,4702],{"class":35,"line":551},[33,4703,4704],{"class":123},"# ─── [Gotcha] Context window ≠ effective recall window ───\n",[33,4706,4707],{"class":35,"line":581},[33,4708,4709],{"class":123},"# A 200K-token context window means the API ACCEPTS 200K tokens — that's all the\n",[33,4711,4712],{"class":35,"line":598},[33,4713,4714],{"class":123},"# guarantee gives you. It does NOT mean the model reliably RETRIEVES a fact from\n",[33,4716,4717],{"class":35,"line":617},[33,4718,4719],{"class":123},"# token position 100,000 in a 200K-token input.\n",[33,4721,4722],{"class":35,"line":631},[33,4723,4724],{"class":123},"# The \"lost in the middle\" effect: recall degrades for mid-context positions.\n",[33,4726,4727],{"class":35,"line":636},[33,4728,4729],{"class":123},"# Production fix: place critical info at the START (primacy) or END (recency) of context.\n",[33,4731,4732],{"class":35,"line":642},[33,4733,4734],{"class":123},"#   system_prompt → [critical constraints here, position 0-800 tokens]\n",[33,4736,4737],{"class":35,"line":665},[33,4738,4739],{"class":123},"#   retrieved_docs → [bulk context, lower recall expected]\n",[33,4741,4742],{"class":35,"line":686},[33,4743,4744],{"class":123},"#   user_message → [the actual task, near the end, high recency recall]\n",[14,4746,4748],{"id":4747},"spot-the-issue","🧠 Spot the Issue",[891,4750,4751],{},"A developer wants a customer-service bot to always end responses with a satisfaction survey link, so they write this system prompt:",[19,4753,4755],{"filename":4754,"language":3273},"bad_survey_prompt.md",[24,4756,4758],{"className":3276,"code":4757,"language":3273,"meta":28,"style":28},"You are a customer service assistant. Help the user with their question.\nAt the very beginning of your response, before anything else, include this\nexact text: \"Thanks for reaching out! Here's your survey link: [link]\".\nThen answer their question below that.\n",[30,4759,4760,4765,4770,4782],{"__ignoreMap":28},[33,4761,4762],{"class":35,"line":36},[33,4763,4764],{"class":43},"You are a customer service assistant. Help the user with their question.\n",[33,4766,4767],{"class":35,"line":47},[33,4768,4769],{"class":43},"At the very beginning of your response, before anything else, include this\n",[33,4771,4772,4775,4779],{"class":35,"line":61},[33,4773,4774],{"class":43},"exact text: \"Thanks for reaching out! Here's your survey link: [",[33,4776,4778],{"class":4777},"sSQSC","link",[33,4780,4781],{"class":43},"]\".\n",[33,4783,4784],{"class":35,"line":74},[33,4785,4786],{"class":43},"Then answer their question below that.\n",[891,4788,4789,4790,4794,4795,4799],{},"The developer tests it and finds that response ",[4791,4792,4793],"strong",{},"quality"," has gotten noticeably worse — the model answers more superficially and sometimes gets facts wrong that it handled fine before. Why, mechanically, would putting the survey link at the ",[4796,4797,4798],"em",{},"start"," cause this?",[4801,4802,4803,4806,4821,4824,4834],"details",{},[4035,4804,4805],{},"Answer",[891,4807,4808,4809,4812,4813,4816,4817,4820],{},"Generation is autoregressive and left-to-right. Forcing the model to emit the survey boilerplate ",[4791,4810,4811],{},"before"," it generates any of the actual answer means every token of the substantive answer is conditioned on a prefix (",[30,4814,4815],{},"\"Thanks for reaching out! Here's your survey link: [link]\"",") that has ",[4791,4818,4819],{},"zero semantic relevance"," to the customer's question.",[891,4822,4823],{},"The model cannot reason about the problem first and then write the boilerplate — it must commit to the boilerplate token sequence first, and only then begin the real answer, with no \"planning\" tokens preceding it. This is especially damaging for questions that benefit from implicit reasoning before the answer (which is most non-trivial questions). You've forced the model to skip the reasoning-adjacent preamble that would normally precede a careful response.",[891,4825,4826,4829,4830,4833],{},[4791,4827,4828],{},"The fix",": fixed boilerplate that doesn't depend on the model's reasoning should go at the ",[4791,4831,4832],{},"end"," of the response — or, better, be appended by your application code after the API call returns, so it never enters the generation path at all.",[19,4835,4837],{"filename":4836,"language":22},"fixed_survey_pattern.py",[24,4838,4840],{"className":26,"code":4839,"language":22,"meta":28,"style":28},"# BAD — boilerplate in the prompt, forced at the START of generation\nSYSTEM_PROMPT_BAD = \"\"\"You are a customer service assistant.\nBefore anything else, output: 'Thanks for reaching out! Survey: [link]'\nThen answer the question.\"\"\"\n\n# GOOD — boilerplate appended in post-processing, model never generates it\nSYSTEM_PROMPT_GOOD = \"\"\"You are a customer service assistant. Answer the user's question thoroughly.\"\"\"\nSURVEY_FOOTER = \"\\n\\n---\\nThanks for reaching out! Here's your survey link: [link]\"\n\ndef build_response(user_question: str, llm_generate) -> str:\n    answer = llm_generate(system=SYSTEM_PROMPT_GOOD, user=user_question)\n    return answer + SURVEY_FOOTER  # model's generation is uncontaminated; boilerplate is deterministic\n",[30,4841,4842,4847,4857,4862,4867,4871,4876,4886,4908,4912,4931,4958],{"__ignoreMap":28},[33,4843,4844],{"class":35,"line":36},[33,4845,4846],{"class":123},"# BAD — boilerplate in the prompt, forced at the START of generation\n",[33,4848,4849,4852,4854],{"class":35,"line":47},[33,4850,4851],{"class":113},"SYSTEM_PROMPT_BAD",[33,4853,117],{"class":39},[33,4855,4856],{"class":103}," \"\"\"You are a customer service assistant.\n",[33,4858,4859],{"class":35,"line":61},[33,4860,4861],{"class":103},"Before anything else, output: 'Thanks for reaching out! Survey: [link]'\n",[33,4863,4864],{"class":35,"line":74},[33,4865,4866],{"class":103},"Then answer the question.\"\"\"\n",[33,4868,4869],{"class":35,"line":81},[33,4870,78],{"emptyLinePlaceholder":77},[33,4872,4873],{"class":35,"line":88},[33,4874,4875],{"class":123},"# GOOD — boilerplate appended in post-processing, model never generates it\n",[33,4877,4878,4881,4883],{"class":35,"line":100},[33,4879,4880],{"class":113},"SYSTEM_PROMPT_GOOD",[33,4882,117],{"class":39},[33,4884,4885],{"class":103}," \"\"\"You are a customer service assistant. Answer the user's question thoroughly.\"\"\"\n",[33,4887,4888,4891,4893,4896,4899,4902,4905],{"class":35,"line":107},[33,4889,4890],{"class":113},"SURVEY_FOOTER",[33,4892,117],{"class":39},[33,4894,4895],{"class":103}," \"",[33,4897,4898],{"class":113},"\\n\\n",[33,4900,4901],{"class":103},"---",[33,4903,4904],{"class":113},"\\n",[33,4906,4907],{"class":103},"Thanks for reaching out! Here's your survey link: [link]\"\n",[33,4909,4910],{"class":35,"line":127},[33,4911,78],{"emptyLinePlaceholder":77},[33,4913,4914,4916,4919,4922,4924,4927,4929],{"class":35,"line":144},[33,4915,210],{"class":39},[33,4917,4918],{"class":84}," build_response",[33,4920,4921],{"class":43},"(user_question: ",[33,4923,181],{"class":113},[33,4925,4926],{"class":43},", llm_generate) -> ",[33,4928,181],{"class":113},[33,4930,97],{"class":43},[33,4932,4933,4936,4938,4941,4944,4946,4948,4950,4953,4955],{"class":35,"line":159},[33,4934,4935],{"class":43},"    answer ",[33,4937,193],{"class":39},[33,4939,4940],{"class":43}," llm_generate(",[33,4942,4943],{"class":503},"system",[33,4945,193],{"class":39},[33,4947,4880],{"class":113},[33,4949,184],{"class":43},[33,4951,4952],{"class":503},"user",[33,4954,193],{"class":39},[33,4956,4957],{"class":43},"user_question)\n",[33,4959,4960,4962,4965,4967,4970],{"class":35,"line":175},[33,4961,718],{"class":39},[33,4963,4964],{"class":43}," answer ",[33,4966,2534],{"class":39},[33,4968,4969],{"class":113}," SURVEY_FOOTER",[33,4971,4972],{"class":123},"  # model's generation is uncontaminated; boilerplate is deterministic\n",[14,4974,4976],{"id":4975},"key-takeaways","Key Takeaways",[19,4978,4980],{"filename":4979,"language":22},"key_takeaways.py",[24,4981,4983],{"className":26,"code":4982,"language":22,"meta":28,"style":28},"\"\"\"\nThe mechanical core of prompt engineering, in code.\n\"\"\"\n\n# 1. An LLM does ONE thing: predict the next token given all prior tokens.\n#    Every capability — reasoning, coding, conversation — is this in a loop.\n#    def llm(tokens): return sample(softmax(model.forward(tokens)))\n#    def generate(prompt): return [llm(prompt + generated_so_far) for _ in range(max_tokens)]\n\n# 2. Prompting is inference-time only. Weights NEVER change.\n#    You are not teaching — you are SELECTING a region of the pre-trained distribution.\n#    training:   weights -= lr * grad(loss(predictions, targets))    # you weren't here\n#    inference:  output = sample(model.forward(your_tokens))          # your only lever\n\n# 3. Tokenization, not characters\u002Fwords, is the model's unit of perception.\n#    \"strawberry\" = 1-2 tokens, not 9 characters → character-counting fails.\n#    \"54321\" = [\"543\",\"21\"] → arithmetic is pattern completion, not digit math.\n#    Non-English text costs 2-3x more tokens for the same semantic content.\n\n# 4. Context window is a BUDGET, not a guarantee of recall.\n#    budget = max_context - response_reserve\n#    if input_tokens > budget: RAISE, don't silently truncate.\n#    recall(position) is NOT uniform — primacy + recency > middle (\"lost in the middle\").\n\n# 5. Generation is left-to-right autoregressive — early tokens are load-bearing.\n#    token[n] is conditioned on tokens[0..n-1], INCLUDING the model's own output.\n#    If the model starts hedging → further hedging becomes more probable.\n#    If the model starts precise → further precision is reinforced.\n#    → Front-load constraints. Put fixed boilerplate at the END (or in post-processing).\n#    → Ensure system + user instructions are ALIGNED, not competing for attention weight.\n",[30,4984,4985,4989,4994,4998,5002,5007,5012,5017,5022,5026,5031,5036,5041,5046,5050,5055,5060,5065,5070,5074,5079,5084,5089,5094,5098,5103,5108,5113,5118,5123],{"__ignoreMap":28},[33,4986,4987],{"class":35,"line":36},[33,4988,1335],{"class":103},[33,4990,4991],{"class":35,"line":47},[33,4992,4993],{"class":103},"The mechanical core of prompt engineering, in code.\n",[33,4995,4996],{"class":35,"line":61},[33,4997,1335],{"class":103},[33,4999,5000],{"class":35,"line":74},[33,5001,78],{"emptyLinePlaceholder":77},[33,5003,5004],{"class":35,"line":81},[33,5005,5006],{"class":123},"# 1. An LLM does ONE thing: predict the next token given all prior tokens.\n",[33,5008,5009],{"class":35,"line":88},[33,5010,5011],{"class":123},"#    Every capability — reasoning, coding, conversation — is this in a loop.\n",[33,5013,5014],{"class":35,"line":100},[33,5015,5016],{"class":123},"#    def llm(tokens): return sample(softmax(model.forward(tokens)))\n",[33,5018,5019],{"class":35,"line":107},[33,5020,5021],{"class":123},"#    def generate(prompt): return [llm(prompt + generated_so_far) for _ in range(max_tokens)]\n",[33,5023,5024],{"class":35,"line":127},[33,5025,78],{"emptyLinePlaceholder":77},[33,5027,5028],{"class":35,"line":144},[33,5029,5030],{"class":123},"# 2. Prompting is inference-time only. Weights NEVER change.\n",[33,5032,5033],{"class":35,"line":159},[33,5034,5035],{"class":123},"#    You are not teaching — you are SELECTING a region of the pre-trained distribution.\n",[33,5037,5038],{"class":35,"line":175},[33,5039,5040],{"class":123},"#    training:   weights -= lr * grad(loss(predictions, targets))    # you weren't here\n",[33,5042,5043],{"class":35,"line":202},[33,5044,5045],{"class":123},"#    inference:  output = sample(model.forward(your_tokens))          # your only lever\n",[33,5047,5048],{"class":35,"line":207},[33,5049,78],{"emptyLinePlaceholder":77},[33,5051,5052],{"class":35,"line":219},[33,5053,5054],{"class":123},"# 3. Tokenization, not characters\u002Fwords, is the model's unit of perception.\n",[33,5056,5057],{"class":35,"line":228},[33,5058,5059],{"class":123},"#    \"strawberry\" = 1-2 tokens, not 9 characters → character-counting fails.\n",[33,5061,5062],{"class":35,"line":237},[33,5063,5064],{"class":123},"#    \"54321\" = [\"543\",\"21\"] → arithmetic is pattern completion, not digit math.\n",[33,5066,5067],{"class":35,"line":243},[33,5068,5069],{"class":123},"#    Non-English text costs 2-3x more tokens for the same semantic content.\n",[33,5071,5072],{"class":35,"line":263},[33,5073,78],{"emptyLinePlaceholder":77},[33,5075,5076],{"class":35,"line":273},[33,5077,5078],{"class":123},"# 4. Context window is a BUDGET, not a guarantee of recall.\n",[33,5080,5081],{"class":35,"line":279},[33,5082,5083],{"class":123},"#    budget = max_context - response_reserve\n",[33,5085,5086],{"class":35,"line":285},[33,5087,5088],{"class":123},"#    if input_tokens > budget: RAISE, don't silently truncate.\n",[33,5090,5091],{"class":35,"line":291},[33,5092,5093],{"class":123},"#    recall(position) is NOT uniform — primacy + recency > middle (\"lost in the middle\").\n",[33,5095,5096],{"class":35,"line":296},[33,5097,78],{"emptyLinePlaceholder":77},[33,5099,5100],{"class":35,"line":308},[33,5101,5102],{"class":123},"# 5. Generation is left-to-right autoregressive — early tokens are load-bearing.\n",[33,5104,5105],{"class":35,"line":322},[33,5106,5107],{"class":123},"#    token[n] is conditioned on tokens[0..n-1], INCLUDING the model's own output.\n",[33,5109,5110],{"class":35,"line":327},[33,5111,5112],{"class":123},"#    If the model starts hedging → further hedging becomes more probable.\n",[33,5114,5115],{"class":35,"line":333},[33,5116,5117],{"class":123},"#    If the model starts precise → further precision is reinforced.\n",[33,5119,5120],{"class":35,"line":339},[33,5121,5122],{"class":123},"#    → Front-load constraints. Put fixed boilerplate at the END (or in post-processing).\n",[33,5124,5125],{"class":35,"line":345},[33,5126,5127],{"class":123},"#    → Ensure system + user instructions are ALIGNED, not competing for attention weight.\n",[5129,5130,5131],"style",{},"html pre.shiki code .ssxIu, html code.shiki .ssxIu{--shiki-default:#24292E;--shiki-github-dark:#E1E4E8}html pre.shiki code .sCrzJ, html code.shiki .sCrzJ{--shiki-default:#E36209;--shiki-github-dark:#FFAB70}html .default .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .github-dark .shiki span {color: var(--shiki-github-dark);background: var(--shiki-github-dark-bg);font-style: var(--shiki-github-dark-font-style);font-weight: var(--shiki-github-dark-font-weight);text-decoration: var(--shiki-github-dark-text-decoration);}html.github-dark .shiki span {color: var(--shiki-github-dark);background: var(--shiki-github-dark-bg);font-style: var(--shiki-github-dark-font-style);font-weight: var(--shiki-github-dark-font-weight);text-decoration: var(--shiki-github-dark-text-decoration);}html pre.shiki code .sSQSC, html code.shiki .sSQSC{--shiki-default:#032F62;--shiki-default-text-decoration:underline;--shiki-github-dark:#DBEDFF;--shiki-github-dark-text-decoration:underline}html pre.shiki code .svdQ7, html code.shiki .svdQ7{--shiki-default:#D73A49;--shiki-github-dark:#F97583}html pre.shiki code .sIsaT, html code.shiki .sIsaT{--shiki-default:#6F42C1;--shiki-github-dark:#B392F0}html pre.shiki code .sJ6F3, html code.shiki .sJ6F3{--shiki-default:#032F62;--shiki-github-dark:#9ECBFF}html pre.shiki code .snvgF, html code.shiki .snvgF{--shiki-default:#005CC5;--shiki-github-dark:#79B8FF}html pre.shiki code .sdCPZ, html code.shiki .sdCPZ{--shiki-default:#6A737D;--shiki-github-dark:#6A737D}",{"title":28,"searchDepth":47,"depth":47,"links":5133},[5134,5135,5136,5137,5138,5139,5140,5141,5142,5143],{"id":16,"depth":47,"text":17},{"id":904,"depth":47,"text":905},{"id":1321,"depth":47,"text":1322},{"id":1953,"depth":47,"text":1954},{"id":2960,"depth":47,"text":2961},{"id":3268,"depth":47,"text":3269},{"id":4082,"depth":47,"text":4083},{"id":4287,"depth":47,"text":4288},{"id":4747,"depth":47,"text":4748},{"id":4975,"depth":47,"text":4976},"The mechanical underpinnings of LLMs — next-token prediction, tokenization, context budgets, and inference-time conditioning — shown through production-grade annotated code, anti-patterns, edge cases, and failure modes.","md",{},"\u002Fprompt-engineering\u002F01-introduction-and-how-llms-work",{"title":5,"description":5144},"prompt-engineering\u002F01-introduction-and-how-llms-work","Z8BdbbEV2DzBBY6RrD8SIkZLUWPB4LsPMl3tG4NniRQ",1789924650775]