[{"data":1,"prerenderedAt":3457},["ShallowReactive",2],{"page-\u002Fpython\u002F06-strings-and-text":3},{"id":4,"title":5,"body":6,"description":27,"extension":3451,"meta":3452,"navigation":51,"path":3453,"seo":3454,"stem":3455,"__hash__":3456},"content\u002Fpython\u002F06-strings-and-text.md","06 — Strings & Text",{"type":7,"value":8,"toc":3429},"minimark",[9,13,18,512,817,821,973,997,1001,1257,1292,1296,1664,1677,1848,1864,1871,1884,2056,2109,2115,2119,2217,2224,2522,2526,2653,2690,2870,2874,2987,2997,3001,3086,3090,3200,3204,3207,3286,3364,3368,3425],[10,11,5],"h1",{"id":12},"_06-strings-text",[14,15,17],"h2",{"id":16},"string-representation-unicode-code-points-not-bytes-or-graphemes","String Representation — Unicode Code Points, Not Bytes or Graphemes",[19,20,22],"code-wrapper",{"language":21},"python",[23,24,28],"pre",{"className":25,"code":26,"language":21,"meta":27,"style":27},"language-python shiki shiki-themes github-light github-dark","# Python str stores Unicode CODE POINTS (integers in 0..0x10FFFF).\n# This is NEITHER bytes (the on-disk encoding) NOR graphemes (what users see).\n\n# ── The three-level model ──\ns = \"café\"\nprint(len(s))                     # 4 — four CODE POINTS: c, a, f, é\nprint(len(s.encode(\"utf-8\")))     # 5 — five BYTES (é is 2 bytes in UTF-8: 0xC3 0xA9)\n# User-perceived \"characters\" (grapheme clusters) can differ from both — see below\n\n# ── Code point ↔ integer conversion ──\nprint(ord(\"A\"))                   # 65 — the integer value of code point U+0041\nprint(chr(65))                    # 'A' — code point U+0041 → str\nprint(hex(ord(\"é\")))             # '0xe9' — U+00E9 (precomposed form)\n\n# ── Production: Unicode normalization before comparison\u002Flookup ──\n# \"é\" has TWO valid Unicode representations:\n#   NFC (precomposed):  U+00E9      — one code point, \"é\"\n#   NFD (decomposed):   U+0065 + U+0301 — \"e\" + combining acute accent — two code points\nimport unicodedata\n\nnfc = \"é\"                          # precomposed — \\u00e9\nnfd = \"e\\u0301\"                    # decomposed — \"e\" + combining accent\n\nprint(nfc == nfd)                  # False! — different code point sequences\nprint(len(nfc), len(nfd))          # 1 2 — different lengths!\nprint(unicodedata.normalize(\"NFC\", nfd) == nfc)   # True — normalize before comparing\n\n# ANTI-PATTERN: using user-supplied strings as dict keys without normalization\n# A user who types \"café\" via a compose key vs. a dead-key accent produces\n# different code point sequences — they'll be DIFFERENT keys in a dict.\nuser_input_a = \"café\"              # NFC from one input method\nuser_input_b = \"cafe\\u0301\"        # NFD from another input method\ncache = {user_input_a: \"data\"}\nprint(user_input_b in cache)       # False — miss! same visual string, different code points\n\n# CORRECT: normalize all strings to NFC before use as keys\u002Fcomparisons\ndef normalize_key(s: str) -> str:\n    return unicodedata.normalize(\"NFC\", s)\n\ncache = {normalize_key(user_input_a): \"data\"}\nprint(normalize_key(user_input_b) in cache)   # True — normalized comparison works\n","",[29,30,31,40,46,53,59,74,93,114,120,125,131,152,173,198,203,209,215,221,227,236,241,255,275,280,297,317,339,344,350,356,362,376,394,411,428,433,439,463,477,482,496],"code",{"__ignoreMap":27},[32,33,36],"span",{"class":34,"line":35},"line",1,[32,37,39],{"class":38},"sdCPZ","# Python str stores Unicode CODE POINTS (integers in 0..0x10FFFF).\n",[32,41,43],{"class":34,"line":42},2,[32,44,45],{"class":38},"# This is NEITHER bytes (the on-disk encoding) NOR graphemes (what users see).\n",[32,47,49],{"class":34,"line":48},3,[32,50,52],{"emptyLinePlaceholder":51},true,"\n",[32,54,56],{"class":34,"line":55},4,[32,57,58],{"class":38},"# ── The three-level model ──\n",[32,60,62,66,70],{"class":34,"line":61},5,[32,63,65],{"class":64},"ssxIu","s ",[32,67,69],{"class":68},"svdQ7","=",[32,71,73],{"class":72},"sJ6F3"," \"café\"\n",[32,75,77,81,84,87,90],{"class":34,"line":76},6,[32,78,80],{"class":79},"snvgF","print",[32,82,83],{"class":64},"(",[32,85,86],{"class":79},"len",[32,88,89],{"class":64},"(s))                     ",[32,91,92],{"class":38},"# 4 — four CODE POINTS: c, a, f, é\n",[32,94,96,98,100,102,105,108,111],{"class":34,"line":95},7,[32,97,80],{"class":79},[32,99,83],{"class":64},[32,101,86],{"class":79},[32,103,104],{"class":64},"(s.encode(",[32,106,107],{"class":72},"\"utf-8\"",[32,109,110],{"class":64},")))     ",[32,112,113],{"class":38},"# 5 — five BYTES (é is 2 bytes in UTF-8: 0xC3 0xA9)\n",[32,115,117],{"class":34,"line":116},8,[32,118,119],{"class":38},"# User-perceived \"characters\" (grapheme clusters) can differ from both — see below\n",[32,121,123],{"class":34,"line":122},9,[32,124,52],{"emptyLinePlaceholder":51},[32,126,128],{"class":34,"line":127},10,[32,129,130],{"class":38},"# ── Code point ↔ integer conversion ──\n",[32,132,134,136,138,141,143,146,149],{"class":34,"line":133},11,[32,135,80],{"class":79},[32,137,83],{"class":64},[32,139,140],{"class":79},"ord",[32,142,83],{"class":64},[32,144,145],{"class":72},"\"A\"",[32,147,148],{"class":64},"))                   ",[32,150,151],{"class":38},"# 65 — the integer value of code point U+0041\n",[32,153,155,157,159,162,164,167,170],{"class":34,"line":154},12,[32,156,80],{"class":79},[32,158,83],{"class":64},[32,160,161],{"class":79},"chr",[32,163,83],{"class":64},[32,165,166],{"class":79},"65",[32,168,169],{"class":64},"))                    ",[32,171,172],{"class":38},"# 'A' — code point U+0041 → str\n",[32,174,176,178,180,183,185,187,189,192,195],{"class":34,"line":175},13,[32,177,80],{"class":79},[32,179,83],{"class":64},[32,181,182],{"class":79},"hex",[32,184,83],{"class":64},[32,186,140],{"class":79},[32,188,83],{"class":64},[32,190,191],{"class":72},"\"é\"",[32,193,194],{"class":64},")))             ",[32,196,197],{"class":38},"# '0xe9' — U+00E9 (precomposed form)\n",[32,199,201],{"class":34,"line":200},14,[32,202,52],{"emptyLinePlaceholder":51},[32,204,206],{"class":34,"line":205},15,[32,207,208],{"class":38},"# ── Production: Unicode normalization before comparison\u002Flookup ──\n",[32,210,212],{"class":34,"line":211},16,[32,213,214],{"class":38},"# \"é\" has TWO valid Unicode representations:\n",[32,216,218],{"class":34,"line":217},17,[32,219,220],{"class":38},"#   NFC (precomposed):  U+00E9      — one code point, \"é\"\n",[32,222,224],{"class":34,"line":223},18,[32,225,226],{"class":38},"#   NFD (decomposed):   U+0065 + U+0301 — \"e\" + combining acute accent — two code points\n",[32,228,230,233],{"class":34,"line":229},19,[32,231,232],{"class":68},"import",[32,234,235],{"class":64}," unicodedata\n",[32,237,239],{"class":34,"line":238},20,[32,240,52],{"emptyLinePlaceholder":51},[32,242,244,247,249,252],{"class":34,"line":243},21,[32,245,246],{"class":64},"nfc ",[32,248,69],{"class":68},[32,250,251],{"class":72}," \"é\"",[32,253,254],{"class":38},"                          # precomposed — \\u00e9\n",[32,256,258,261,263,266,269,272],{"class":34,"line":257},22,[32,259,260],{"class":64},"nfd ",[32,262,69],{"class":68},[32,264,265],{"class":72}," \"e",[32,267,268],{"class":79},"\\u0301",[32,270,271],{"class":72},"\"",[32,273,274],{"class":38},"                    # decomposed — \"e\" + combining accent\n",[32,276,278],{"class":34,"line":277},23,[32,279,52],{"emptyLinePlaceholder":51},[32,281,283,285,288,291,294],{"class":34,"line":282},24,[32,284,80],{"class":79},[32,286,287],{"class":64},"(nfc ",[32,289,290],{"class":68},"==",[32,292,293],{"class":64}," nfd)                  ",[32,295,296],{"class":38},"# False! — different code point sequences\n",[32,298,300,302,304,306,309,311,314],{"class":34,"line":299},25,[32,301,80],{"class":79},[32,303,83],{"class":64},[32,305,86],{"class":79},[32,307,308],{"class":64},"(nfc), ",[32,310,86],{"class":79},[32,312,313],{"class":64},"(nfd))          ",[32,315,316],{"class":38},"# 1 2 — different lengths!\n",[32,318,320,322,325,328,331,333,336],{"class":34,"line":319},26,[32,321,80],{"class":79},[32,323,324],{"class":64},"(unicodedata.normalize(",[32,326,327],{"class":72},"\"NFC\"",[32,329,330],{"class":64},", nfd) ",[32,332,290],{"class":68},[32,334,335],{"class":64}," nfc)   ",[32,337,338],{"class":38},"# True — normalize before comparing\n",[32,340,342],{"class":34,"line":341},27,[32,343,52],{"emptyLinePlaceholder":51},[32,345,347],{"class":34,"line":346},28,[32,348,349],{"class":38},"# ANTI-PATTERN: using user-supplied strings as dict keys without normalization\n",[32,351,353],{"class":34,"line":352},29,[32,354,355],{"class":38},"# A user who types \"café\" via a compose key vs. a dead-key accent produces\n",[32,357,359],{"class":34,"line":358},30,[32,360,361],{"class":38},"# different code point sequences — they'll be DIFFERENT keys in a dict.\n",[32,363,365,368,370,373],{"class":34,"line":364},31,[32,366,367],{"class":64},"user_input_a ",[32,369,69],{"class":68},[32,371,372],{"class":72}," \"café\"",[32,374,375],{"class":38},"              # NFC from one input method\n",[32,377,379,382,384,387,389,391],{"class":34,"line":378},32,[32,380,381],{"class":64},"user_input_b ",[32,383,69],{"class":68},[32,385,386],{"class":72}," \"cafe",[32,388,268],{"class":79},[32,390,271],{"class":72},[32,392,393],{"class":38},"        # NFD from another input method\n",[32,395,397,400,402,405,408],{"class":34,"line":396},33,[32,398,399],{"class":64},"cache ",[32,401,69],{"class":68},[32,403,404],{"class":64}," {user_input_a: ",[32,406,407],{"class":72},"\"data\"",[32,409,410],{"class":64},"}\n",[32,412,414,416,419,422,425],{"class":34,"line":413},34,[32,415,80],{"class":79},[32,417,418],{"class":64},"(user_input_b ",[32,420,421],{"class":68},"in",[32,423,424],{"class":64}," cache)       ",[32,426,427],{"class":38},"# False — miss! same visual string, different code points\n",[32,429,431],{"class":34,"line":430},35,[32,432,52],{"emptyLinePlaceholder":51},[32,434,436],{"class":34,"line":435},36,[32,437,438],{"class":38},"# CORRECT: normalize all strings to NFC before use as keys\u002Fcomparisons\n",[32,440,442,445,449,452,455,458,460],{"class":34,"line":441},37,[32,443,444],{"class":68},"def",[32,446,448],{"class":447},"sIsaT"," normalize_key",[32,450,451],{"class":64},"(s: ",[32,453,454],{"class":79},"str",[32,456,457],{"class":64},") -> ",[32,459,454],{"class":79},[32,461,462],{"class":64},":\n",[32,464,466,469,472,474],{"class":34,"line":465},38,[32,467,468],{"class":68},"    return",[32,470,471],{"class":64}," unicodedata.normalize(",[32,473,327],{"class":72},[32,475,476],{"class":64},", s)\n",[32,478,480],{"class":34,"line":479},39,[32,481,52],{"emptyLinePlaceholder":51},[32,483,485,487,489,492,494],{"class":34,"line":484},40,[32,486,399],{"class":64},[32,488,69],{"class":68},[32,490,491],{"class":64}," {normalize_key(user_input_a): ",[32,493,407],{"class":72},[32,495,410],{"class":64},[32,497,499,501,504,506,509],{"class":34,"line":498},41,[32,500,80],{"class":79},[32,502,503],{"class":64},"(normalize_key(user_input_b) ",[32,505,421],{"class":68},[32,507,508],{"class":64}," cache)   ",[32,510,511],{"class":38},"# True — normalized comparison works\n",[19,513,514],{"language":21},[23,515,517],{"className":25,"code":516,"language":21,"meta":27,"style":27},"# ── Grapheme clusters: what users perceive vs. what len() reports ──\n# Emoji and flags are the most common source of \"len() is wrong\" bugs.\n\nflag = \"🇺🇸\"                       # US flag — two REGIONAL INDICATOR code points\nprint(len(flag))                   # 2 — two code points, one visible glyph\nprint([hex(ord(c)) for c in flag]) # ['0x1f1fa', '0x1f1f8'] — U=U+1F1FA, S=U+1F1F8\n\nfamily = \"👨‍👩‍👧‍👦\"                  # family emoji — joined by ZWJ (zero-width joiner)\nprint(len(family))                 # 7 — seven code points (4 people + 3 ZWJ), one glyph\n# ZWJ = U+200D — invisible joiner that tells the renderer to combine adjacent emoji\n\n# For correct user-perceived character counting, use the \\X regex grapheme cluster\n# (requires the `regex` module, not stdlib `re`, which lacks grapheme support)\n# import regex\n# print(len(regex.findall(r\"\\X\", family)))   # 1 — one grapheme cluster\n\n# ── Production: truncating user-display text at grapheme boundaries ──\n# Naive truncation at code point N can split a combining character from its base\ndef truncate_naive(s: str, max_len: int) -> str:\n    \"\"\"WRONG — can split combining accents from their base characters.\"\"\"\n    return s[:max_len]              # cuts at code point boundary, not grapheme boundary\n\ndef truncate_safe(s: str, max_len: int) -> str:\n    \"\"\"CORRECT — counts grapheme clusters, not code points (requires `regex` module).\"\"\"\n    import regex\n    clusters = regex.findall(r\"\\X\", s)\n    return \"\".join(clusters[:max_len])\n\n# \"café\" in NFD form: c, a, f, e, combining-acute — truncating at 4 drops the accent\nnfd_cafe = \"cafe\\u0301\"\nprint(truncate_naive(nfd_cafe, 4))   # \"cafe\" — accent lost! displays as \"cafe\" not \"café\"\n# print(truncate_safe(nfd_cafe, 4))  # \"café\" — correct, keeps the combining cluster intact\n",[29,518,519,524,529,533,546,560,590,594,607,621,626,630,635,640,645,650,654,659,664,687,692,702,706,727,732,740,763,773,777,782,796,812],{"__ignoreMap":27},[32,520,521],{"class":34,"line":35},[32,522,523],{"class":38},"# ── Grapheme clusters: what users perceive vs. what len() reports ──\n",[32,525,526],{"class":34,"line":42},[32,527,528],{"class":38},"# Emoji and flags are the most common source of \"len() is wrong\" bugs.\n",[32,530,531],{"class":34,"line":48},[32,532,52],{"emptyLinePlaceholder":51},[32,534,535,538,540,543],{"class":34,"line":55},[32,536,537],{"class":64},"flag ",[32,539,69],{"class":68},[32,541,542],{"class":72}," \"🇺🇸\"",[32,544,545],{"class":38},"                       # US flag — two REGIONAL INDICATOR code points\n",[32,547,548,550,552,554,557],{"class":34,"line":61},[32,549,80],{"class":79},[32,551,83],{"class":64},[32,553,86],{"class":79},[32,555,556],{"class":64},"(flag))                   ",[32,558,559],{"class":38},"# 2 — two code points, one visible glyph\n",[32,561,562,564,567,569,571,573,576,579,582,584,587],{"class":34,"line":76},[32,563,80],{"class":79},[32,565,566],{"class":64},"([",[32,568,182],{"class":79},[32,570,83],{"class":64},[32,572,140],{"class":79},[32,574,575],{"class":64},"(c)) ",[32,577,578],{"class":68},"for",[32,580,581],{"class":64}," c ",[32,583,421],{"class":68},[32,585,586],{"class":64}," flag]) ",[32,588,589],{"class":38},"# ['0x1f1fa', '0x1f1f8'] — U=U+1F1FA, S=U+1F1F8\n",[32,591,592],{"class":34,"line":95},[32,593,52],{"emptyLinePlaceholder":51},[32,595,596,599,601,604],{"class":34,"line":116},[32,597,598],{"class":64},"family ",[32,600,69],{"class":68},[32,602,603],{"class":72}," \"👨‍👩‍👧‍👦\"",[32,605,606],{"class":38},"                  # family emoji — joined by ZWJ (zero-width joiner)\n",[32,608,609,611,613,615,618],{"class":34,"line":122},[32,610,80],{"class":79},[32,612,83],{"class":64},[32,614,86],{"class":79},[32,616,617],{"class":64},"(family))                 ",[32,619,620],{"class":38},"# 7 — seven code points (4 people + 3 ZWJ), one glyph\n",[32,622,623],{"class":34,"line":127},[32,624,625],{"class":38},"# ZWJ = U+200D — invisible joiner that tells the renderer to combine adjacent emoji\n",[32,627,628],{"class":34,"line":133},[32,629,52],{"emptyLinePlaceholder":51},[32,631,632],{"class":34,"line":154},[32,633,634],{"class":38},"# For correct user-perceived character counting, use the \\X regex grapheme cluster\n",[32,636,637],{"class":34,"line":175},[32,638,639],{"class":38},"# (requires the `regex` module, not stdlib `re`, which lacks grapheme support)\n",[32,641,642],{"class":34,"line":200},[32,643,644],{"class":38},"# import regex\n",[32,646,647],{"class":34,"line":205},[32,648,649],{"class":38},"# print(len(regex.findall(r\"\\X\", family)))   # 1 — one grapheme cluster\n",[32,651,652],{"class":34,"line":211},[32,653,52],{"emptyLinePlaceholder":51},[32,655,656],{"class":34,"line":217},[32,657,658],{"class":38},"# ── Production: truncating user-display text at grapheme boundaries ──\n",[32,660,661],{"class":34,"line":223},[32,662,663],{"class":38},"# Naive truncation at code point N can split a combining character from its base\n",[32,665,666,668,671,673,675,678,681,683,685],{"class":34,"line":229},[32,667,444],{"class":68},[32,669,670],{"class":447}," truncate_naive",[32,672,451],{"class":64},[32,674,454],{"class":79},[32,676,677],{"class":64},", max_len: ",[32,679,680],{"class":79},"int",[32,682,457],{"class":64},[32,684,454],{"class":79},[32,686,462],{"class":64},[32,688,689],{"class":34,"line":238},[32,690,691],{"class":72},"    \"\"\"WRONG — can split combining accents from their base characters.\"\"\"\n",[32,693,694,696,699],{"class":34,"line":243},[32,695,468],{"class":68},[32,697,698],{"class":64}," s[:max_len]              ",[32,700,701],{"class":38},"# cuts at code point boundary, not grapheme boundary\n",[32,703,704],{"class":34,"line":257},[32,705,52],{"emptyLinePlaceholder":51},[32,707,708,710,713,715,717,719,721,723,725],{"class":34,"line":277},[32,709,444],{"class":68},[32,711,712],{"class":447}," truncate_safe",[32,714,451],{"class":64},[32,716,454],{"class":79},[32,718,677],{"class":64},[32,720,680],{"class":79},[32,722,457],{"class":64},[32,724,454],{"class":79},[32,726,462],{"class":64},[32,728,729],{"class":34,"line":282},[32,730,731],{"class":72},"    \"\"\"CORRECT — counts grapheme clusters, not code points (requires `regex` module).\"\"\"\n",[32,733,734,737],{"class":34,"line":299},[32,735,736],{"class":68},"    import",[32,738,739],{"class":64}," regex\n",[32,741,742,745,747,750,753,755,759,761],{"class":34,"line":319},[32,743,744],{"class":64},"    clusters ",[32,746,69],{"class":68},[32,748,749],{"class":64}," regex.findall(",[32,751,752],{"class":68},"r",[32,754,271],{"class":72},[32,756,758],{"class":757},"snRuI","\\X",[32,760,271],{"class":72},[32,762,476],{"class":64},[32,764,765,767,770],{"class":34,"line":341},[32,766,468],{"class":68},[32,768,769],{"class":72}," \"\"",[32,771,772],{"class":64},".join(clusters[:max_len])\n",[32,774,775],{"class":34,"line":346},[32,776,52],{"emptyLinePlaceholder":51},[32,778,779],{"class":34,"line":352},[32,780,781],{"class":38},"# \"café\" in NFD form: c, a, f, e, combining-acute — truncating at 4 drops the accent\n",[32,783,784,787,789,791,793],{"class":34,"line":358},[32,785,786],{"class":64},"nfd_cafe ",[32,788,69],{"class":68},[32,790,386],{"class":72},[32,792,268],{"class":79},[32,794,795],{"class":72},"\"\n",[32,797,798,800,803,806,809],{"class":34,"line":364},[32,799,80],{"class":79},[32,801,802],{"class":64},"(truncate_naive(nfd_cafe, ",[32,804,805],{"class":79},"4",[32,807,808],{"class":64},"))   ",[32,810,811],{"class":38},"# \"cafe\" — accent lost! displays as \"cafe\" not \"café\"\n",[32,813,814],{"class":34,"line":378},[32,815,816],{"class":38},"# print(truncate_safe(nfd_cafe, 4))  # \"café\" — correct, keeps the combining cluster intact\n",[14,818,820],{"id":819},"indexing-and-slicing","Indexing and Slicing",[19,822,823],{"language":21},[23,824,826],{"className":25,"code":825,"language":21,"meta":27,"style":27},"s = \"Hello, World!\"\nprint(s[0])          # 'H'\nprint(s[-1])           # '!'          — negative indices count from the end\nprint(s[7:12])           # 'World'\nprint(s[:5])                # 'Hello'\nprint(s[7:])                  # 'World!'\nprint(s[::-1])                  # '!dlroW ,olleH'  — reverse via step -1\nprint(s[::2])                     # 'Hlo ol!'        — every other character\nprint(s[100:200])                   # ''  — out-of-range slices never raise, just return empty\u002Fpartial\n",[29,827,828,837,853,871,891,907,921,938,953],{"__ignoreMap":27},[32,829,830,832,834],{"class":34,"line":35},[32,831,65],{"class":64},[32,833,69],{"class":68},[32,835,836],{"class":72}," \"Hello, World!\"\n",[32,838,839,841,844,847,850],{"class":34,"line":42},[32,840,80],{"class":79},[32,842,843],{"class":64},"(s[",[32,845,846],{"class":79},"0",[32,848,849],{"class":64},"])          ",[32,851,852],{"class":38},"# 'H'\n",[32,854,855,857,859,862,865,868],{"class":34,"line":48},[32,856,80],{"class":79},[32,858,843],{"class":64},[32,860,861],{"class":68},"-",[32,863,864],{"class":79},"1",[32,866,867],{"class":64},"])           ",[32,869,870],{"class":38},"# '!'          — negative indices count from the end\n",[32,872,873,875,877,880,883,886,888],{"class":34,"line":55},[32,874,80],{"class":79},[32,876,843],{"class":64},[32,878,879],{"class":79},"7",[32,881,882],{"class":64},":",[32,884,885],{"class":79},"12",[32,887,867],{"class":64},[32,889,890],{"class":38},"# 'World'\n",[32,892,893,895,898,901,904],{"class":34,"line":61},[32,894,80],{"class":79},[32,896,897],{"class":64},"(s[:",[32,899,900],{"class":79},"5",[32,902,903],{"class":64},"])                ",[32,905,906],{"class":38},"# 'Hello'\n",[32,908,909,911,913,915,918],{"class":34,"line":76},[32,910,80],{"class":79},[32,912,843],{"class":64},[32,914,879],{"class":79},[32,916,917],{"class":64},":])                  ",[32,919,920],{"class":38},"# 'World!'\n",[32,922,923,925,928,930,932,935],{"class":34,"line":95},[32,924,80],{"class":79},[32,926,927],{"class":64},"(s[::",[32,929,861],{"class":68},[32,931,864],{"class":79},[32,933,934],{"class":64},"])                  ",[32,936,937],{"class":38},"# '!dlroW ,olleH'  — reverse via step -1\n",[32,939,940,942,944,947,950],{"class":34,"line":116},[32,941,80],{"class":79},[32,943,927],{"class":64},[32,945,946],{"class":79},"2",[32,948,949],{"class":64},"])                     ",[32,951,952],{"class":38},"# 'Hlo ol!'        — every other character\n",[32,954,955,957,959,962,964,967,970],{"class":34,"line":122},[32,956,80],{"class":79},[32,958,843],{"class":64},[32,960,961],{"class":79},"100",[32,963,882],{"class":64},[32,965,966],{"class":79},"200",[32,968,969],{"class":64},"])                   ",[32,971,972],{"class":38},"# ''  — out-of-range slices never raise, just return empty\u002Fpartial\n",[974,975,976,980,981,984,985,988,989,993,994,996],"p",{},[977,978,979],"strong",{},"Best practice",": slicing never raises ",[29,982,983],{},"IndexError",", even for wildly out-of-range bounds — it clamps silently. Direct indexing (",[29,986,987],{},"s[100]",") ",[990,991,992],"em",{},"does"," raise ",[29,995,983],{},". This asymmetry is worth internalizing: if you need bounds-safety, prefer slicing or explicit length checks over bare indexing.",[14,998,1000],{"id":999},"essential-string-methods","Essential String Methods",[19,1002,1003],{"language":21},[23,1004,1006],{"className":25,"code":1005,"language":21,"meta":27,"style":27},"s = \"  Hello, World!  \"\n\nprint(s.strip())            # \"Hello, World!\"      — trims whitespace both sides\nprint(s.lower())               # \"  hello, world!  \"\nprint(s.upper())                  # \"  HELLO, WORLD!  \"\nprint(s.replace(\"World\", \"Python\"))  # \"  Hello, Python!  \"\nprint(s.strip().split(\", \"))            # ['Hello', 'World!']\nprint(\"-\".join([\"a\", \"b\", \"c\"]))          # \"a-b-c\"\nprint(\"Hello\".startswith(\"He\"))              # True\nprint(\"Hello\".endswith(\"lo\"))                  # True\nprint(\"Hello\".find(\"l\"))                         # 2  — index of first match, -1 if absent\nprint(\"Hello\".index(\"z\"))                          # ValueError: substring not found (raises!)\nprint(\"hello world\".title())                          # \"Hello World\"\nprint(\"  \".isspace())                                    # True\nprint(\"42\".isdigit())                                      # True\nprint(\"Hello123\".isalnum())                                   # True\n",[29,1007,1008,1017,1021,1031,1041,1051,1073,1089,1120,1141,1160,1180,1200,1215,1229,1243],{"__ignoreMap":27},[32,1009,1010,1012,1014],{"class":34,"line":35},[32,1011,65],{"class":64},[32,1013,69],{"class":68},[32,1015,1016],{"class":72}," \"  Hello, World!  \"\n",[32,1018,1019],{"class":34,"line":42},[32,1020,52],{"emptyLinePlaceholder":51},[32,1022,1023,1025,1028],{"class":34,"line":48},[32,1024,80],{"class":79},[32,1026,1027],{"class":64},"(s.strip())            ",[32,1029,1030],{"class":38},"# \"Hello, World!\"      — trims whitespace both sides\n",[32,1032,1033,1035,1038],{"class":34,"line":55},[32,1034,80],{"class":79},[32,1036,1037],{"class":64},"(s.lower())               ",[32,1039,1040],{"class":38},"# \"  hello, world!  \"\n",[32,1042,1043,1045,1048],{"class":34,"line":61},[32,1044,80],{"class":79},[32,1046,1047],{"class":64},"(s.upper())                  ",[32,1049,1050],{"class":38},"# \"  HELLO, WORLD!  \"\n",[32,1052,1053,1055,1058,1061,1064,1067,1070],{"class":34,"line":76},[32,1054,80],{"class":79},[32,1056,1057],{"class":64},"(s.replace(",[32,1059,1060],{"class":72},"\"World\"",[32,1062,1063],{"class":64},", ",[32,1065,1066],{"class":72},"\"Python\"",[32,1068,1069],{"class":64},"))  ",[32,1071,1072],{"class":38},"# \"  Hello, Python!  \"\n",[32,1074,1075,1077,1080,1083,1086],{"class":34,"line":95},[32,1076,80],{"class":79},[32,1078,1079],{"class":64},"(s.strip().split(",[32,1081,1082],{"class":72},"\", \"",[32,1084,1085],{"class":64},"))            ",[32,1087,1088],{"class":38},"# ['Hello', 'World!']\n",[32,1090,1091,1093,1095,1098,1101,1104,1106,1109,1111,1114,1117],{"class":34,"line":116},[32,1092,80],{"class":79},[32,1094,83],{"class":64},[32,1096,1097],{"class":72},"\"-\"",[32,1099,1100],{"class":64},".join([",[32,1102,1103],{"class":72},"\"a\"",[32,1105,1063],{"class":64},[32,1107,1108],{"class":72},"\"b\"",[32,1110,1063],{"class":64},[32,1112,1113],{"class":72},"\"c\"",[32,1115,1116],{"class":64},"]))          ",[32,1118,1119],{"class":38},"# \"a-b-c\"\n",[32,1121,1122,1124,1126,1129,1132,1135,1138],{"class":34,"line":122},[32,1123,80],{"class":79},[32,1125,83],{"class":64},[32,1127,1128],{"class":72},"\"Hello\"",[32,1130,1131],{"class":64},".startswith(",[32,1133,1134],{"class":72},"\"He\"",[32,1136,1137],{"class":64},"))              ",[32,1139,1140],{"class":38},"# True\n",[32,1142,1143,1145,1147,1149,1152,1155,1158],{"class":34,"line":127},[32,1144,80],{"class":79},[32,1146,83],{"class":64},[32,1148,1128],{"class":72},[32,1150,1151],{"class":64},".endswith(",[32,1153,1154],{"class":72},"\"lo\"",[32,1156,1157],{"class":64},"))                  ",[32,1159,1140],{"class":38},[32,1161,1162,1164,1166,1168,1171,1174,1177],{"class":34,"line":133},[32,1163,80],{"class":79},[32,1165,83],{"class":64},[32,1167,1128],{"class":72},[32,1169,1170],{"class":64},".find(",[32,1172,1173],{"class":72},"\"l\"",[32,1175,1176],{"class":64},"))                         ",[32,1178,1179],{"class":38},"# 2  — index of first match, -1 if absent\n",[32,1181,1182,1184,1186,1188,1191,1194,1197],{"class":34,"line":154},[32,1183,80],{"class":79},[32,1185,83],{"class":64},[32,1187,1128],{"class":72},[32,1189,1190],{"class":64},".index(",[32,1192,1193],{"class":72},"\"z\"",[32,1195,1196],{"class":64},"))                          ",[32,1198,1199],{"class":38},"# ValueError: substring not found (raises!)\n",[32,1201,1202,1204,1206,1209,1212],{"class":34,"line":175},[32,1203,80],{"class":79},[32,1205,83],{"class":64},[32,1207,1208],{"class":72},"\"hello world\"",[32,1210,1211],{"class":64},".title())                          ",[32,1213,1214],{"class":38},"# \"Hello World\"\n",[32,1216,1217,1219,1221,1224,1227],{"class":34,"line":200},[32,1218,80],{"class":79},[32,1220,83],{"class":64},[32,1222,1223],{"class":72},"\"  \"",[32,1225,1226],{"class":64},".isspace())                                    ",[32,1228,1140],{"class":38},[32,1230,1231,1233,1235,1238,1241],{"class":34,"line":205},[32,1232,80],{"class":79},[32,1234,83],{"class":64},[32,1236,1237],{"class":72},"\"42\"",[32,1239,1240],{"class":64},".isdigit())                                      ",[32,1242,1140],{"class":38},[32,1244,1245,1247,1249,1252,1255],{"class":34,"line":211},[32,1246,80],{"class":79},[32,1248,83],{"class":64},[32,1250,1251],{"class":72},"\"Hello123\"",[32,1253,1254],{"class":64},".isalnum())                                   ",[32,1256,1140],{"class":38},[974,1258,1259,1262,1263,1266,1267,1270,1271,1274,1275,1277,1278,1281,1282,1285,1286,1277,1288,1291],{},[29,1260,1261],{},".find()"," returns ",[29,1264,1265],{},"-1"," on failure; ",[29,1268,1269],{},".index()"," raises ",[29,1272,1273],{},"ValueError"," on failure. Choosing between them is choosing between EAFP (",[29,1276,1269],{}," + ",[29,1279,1280],{},"try","\u002F",[29,1283,1284],{},"except",") and a manual sentinel check (",[29,1287,1261],{},[29,1289,1290],{},"if result == -1",").",[14,1293,1295],{"id":1294},"f-strings-the-modern-standard-for-formatting","f-strings — The Modern Standard for Formatting",[19,1297,1298],{"language":21},[23,1299,1301],{"className":25,"code":1300,"language":21,"meta":27,"style":27},"name = \"Ada\"\nage = 36\npi = 3.14159265\n\nprint(f\"{name} is {age} years old\")             # Ada is 36 years old\nprint(f\"{name!r}\")                                 # 'Ada'  — !r calls repr()\nprint(f\"{pi:.2f}\")                                    # 3.14  — format spec: 2 decimal places\nprint(f\"{1234567:,}\")                                    # 1,234,567 — thousands separator\nprint(f\"{age:>5}\")                                          # \"   36\" — right-aligned, width 5\nprint(f\"{age:\u003C5}|\")                                            # \"36   |\" — left-aligned\nprint(f\"{age:^5}|\")                                                # \" 36  |\" — centered\n\n# Self-documenting expressions (3.8+) — the = specifier\nprint(f\"{name=}\")                                                     # name='Ada'\nprint(f\"{age * 2=}\")                                                    # age * 2=72\n\n# Nested expressions and method calls work directly inside braces\nitems = [\"apple\", \"banana\"]\nprint(f\"Items: {', '.join(items).upper()}\")                              # Items: APPLE, BANANA\n",[29,1302,1303,1313,1323,1333,1337,1376,1403,1431,1456,1483,1511,1538,1542,1547,1573,1605,1609,1614,1635],{"__ignoreMap":27},[32,1304,1305,1308,1310],{"class":34,"line":35},[32,1306,1307],{"class":64},"name ",[32,1309,69],{"class":68},[32,1311,1312],{"class":72}," \"Ada\"\n",[32,1314,1315,1318,1320],{"class":34,"line":42},[32,1316,1317],{"class":64},"age ",[32,1319,69],{"class":68},[32,1321,1322],{"class":79}," 36\n",[32,1324,1325,1328,1330],{"class":34,"line":48},[32,1326,1327],{"class":64},"pi ",[32,1329,69],{"class":68},[32,1331,1332],{"class":79}," 3.14159265\n",[32,1334,1335],{"class":34,"line":55},[32,1336,52],{"emptyLinePlaceholder":51},[32,1338,1339,1341,1343,1346,1348,1351,1354,1357,1360,1362,1365,1367,1370,1373],{"class":34,"line":61},[32,1340,80],{"class":79},[32,1342,83],{"class":64},[32,1344,1345],{"class":68},"f",[32,1347,271],{"class":72},[32,1349,1350],{"class":79},"{",[32,1352,1353],{"class":64},"name",[32,1355,1356],{"class":79},"}",[32,1358,1359],{"class":72}," is ",[32,1361,1350],{"class":79},[32,1363,1364],{"class":64},"age",[32,1366,1356],{"class":79},[32,1368,1369],{"class":72}," years old\"",[32,1371,1372],{"class":64},")             ",[32,1374,1375],{"class":38},"# Ada is 36 years old\n",[32,1377,1378,1380,1382,1384,1386,1388,1390,1393,1395,1397,1400],{"class":34,"line":76},[32,1379,80],{"class":79},[32,1381,83],{"class":64},[32,1383,1345],{"class":68},[32,1385,271],{"class":72},[32,1387,1350],{"class":79},[32,1389,1353],{"class":64},[32,1391,1392],{"class":68},"!r",[32,1394,1356],{"class":79},[32,1396,271],{"class":72},[32,1398,1399],{"class":64},")                                 ",[32,1401,1402],{"class":38},"# 'Ada'  — !r calls repr()\n",[32,1404,1405,1407,1409,1411,1413,1415,1418,1421,1423,1425,1428],{"class":34,"line":95},[32,1406,80],{"class":79},[32,1408,83],{"class":64},[32,1410,1345],{"class":68},[32,1412,271],{"class":72},[32,1414,1350],{"class":79},[32,1416,1417],{"class":64},"pi",[32,1419,1420],{"class":68},":.2f",[32,1422,1356],{"class":79},[32,1424,271],{"class":72},[32,1426,1427],{"class":64},")                                    ",[32,1429,1430],{"class":38},"# 3.14  — format spec: 2 decimal places\n",[32,1432,1433,1435,1437,1439,1441,1444,1447,1449,1451,1453],{"class":34,"line":116},[32,1434,80],{"class":79},[32,1436,83],{"class":64},[32,1438,1345],{"class":68},[32,1440,271],{"class":72},[32,1442,1443],{"class":79},"{1234567",[32,1445,1446],{"class":68},":,",[32,1448,1356],{"class":79},[32,1450,271],{"class":72},[32,1452,1427],{"class":64},[32,1454,1455],{"class":38},"# 1,234,567 — thousands separator\n",[32,1457,1458,1460,1462,1464,1466,1468,1470,1473,1475,1477,1480],{"class":34,"line":122},[32,1459,80],{"class":79},[32,1461,83],{"class":64},[32,1463,1345],{"class":68},[32,1465,271],{"class":72},[32,1467,1350],{"class":79},[32,1469,1364],{"class":64},[32,1471,1472],{"class":68},":>5",[32,1474,1356],{"class":79},[32,1476,271],{"class":72},[32,1478,1479],{"class":64},")                                          ",[32,1481,1482],{"class":38},"# \"   36\" — right-aligned, width 5\n",[32,1484,1485,1487,1489,1491,1493,1495,1497,1500,1502,1505,1508],{"class":34,"line":127},[32,1486,80],{"class":79},[32,1488,83],{"class":64},[32,1490,1345],{"class":68},[32,1492,271],{"class":72},[32,1494,1350],{"class":79},[32,1496,1364],{"class":64},[32,1498,1499],{"class":68},":\u003C5",[32,1501,1356],{"class":79},[32,1503,1504],{"class":72},"|\"",[32,1506,1507],{"class":64},")                                            ",[32,1509,1510],{"class":38},"# \"36   |\" — left-aligned\n",[32,1512,1513,1515,1517,1519,1521,1523,1525,1528,1530,1532,1535],{"class":34,"line":133},[32,1514,80],{"class":79},[32,1516,83],{"class":64},[32,1518,1345],{"class":68},[32,1520,271],{"class":72},[32,1522,1350],{"class":79},[32,1524,1364],{"class":64},[32,1526,1527],{"class":68},":^5",[32,1529,1356],{"class":79},[32,1531,1504],{"class":72},[32,1533,1534],{"class":64},")                                                ",[32,1536,1537],{"class":38},"# \" 36  |\" — centered\n",[32,1539,1540],{"class":34,"line":154},[32,1541,52],{"emptyLinePlaceholder":51},[32,1543,1544],{"class":34,"line":175},[32,1545,1546],{"class":38},"# Self-documenting expressions (3.8+) — the = specifier\n",[32,1548,1549,1551,1553,1555,1557,1559,1561,1563,1565,1567,1570],{"class":34,"line":200},[32,1550,80],{"class":79},[32,1552,83],{"class":64},[32,1554,1345],{"class":68},[32,1556,271],{"class":72},[32,1558,1350],{"class":79},[32,1560,1353],{"class":64},[32,1562,69],{"class":68},[32,1564,1356],{"class":79},[32,1566,271],{"class":72},[32,1568,1569],{"class":64},")                                                     ",[32,1571,1572],{"class":38},"# name='Ada'\n",[32,1574,1575,1577,1579,1581,1583,1585,1587,1590,1593,1595,1597,1599,1602],{"class":34,"line":205},[32,1576,80],{"class":79},[32,1578,83],{"class":64},[32,1580,1345],{"class":68},[32,1582,271],{"class":72},[32,1584,1350],{"class":79},[32,1586,1317],{"class":64},[32,1588,1589],{"class":68},"*",[32,1591,1592],{"class":79}," 2",[32,1594,69],{"class":68},[32,1596,1356],{"class":79},[32,1598,271],{"class":72},[32,1600,1601],{"class":64},")                                                    ",[32,1603,1604],{"class":38},"# age * 2=72\n",[32,1606,1607],{"class":34,"line":211},[32,1608,52],{"emptyLinePlaceholder":51},[32,1610,1611],{"class":34,"line":217},[32,1612,1613],{"class":38},"# Nested expressions and method calls work directly inside braces\n",[32,1615,1616,1619,1621,1624,1627,1629,1632],{"class":34,"line":223},[32,1617,1618],{"class":64},"items ",[32,1620,69],{"class":68},[32,1622,1623],{"class":64}," [",[32,1625,1626],{"class":72},"\"apple\"",[32,1628,1063],{"class":64},[32,1630,1631],{"class":72},"\"banana\"",[32,1633,1634],{"class":64},"]\n",[32,1636,1637,1639,1641,1643,1646,1648,1651,1654,1656,1658,1661],{"class":34,"line":229},[32,1638,80],{"class":79},[32,1640,83],{"class":64},[32,1642,1345],{"class":68},[32,1644,1645],{"class":72},"\"Items: ",[32,1647,1350],{"class":79},[32,1649,1650],{"class":72},"', '",[32,1652,1653],{"class":64},".join(items).upper()",[32,1655,1356],{"class":79},[32,1657,271],{"class":72},[32,1659,1660],{"class":64},")                              ",[32,1662,1663],{"class":38},"# Items: APPLE, BANANA\n",[1665,1666,1668,1669,1672,1673,1676],"h3",{"id":1667},"f-strings-vs-format-vs-know-all-three-prefer-f-strings","f-strings vs ",[29,1670,1671],{},".format()"," vs ",[29,1674,1675],{},"%"," — know all three, prefer f-strings",[19,1678,1679],{"language":21},[23,1680,1682],{"className":25,"code":1681,"language":21,"meta":27,"style":27},"name, score = \"Ada\", 98\n\n# %-formatting — legacy, still seen in logging calls and old codebases\nprint(\"%s scored %d\" % (name, score))\n\n# .format() — flexible, verbose, used when the template is data (not a literal)\nprint(\"{} scored {}\".format(name, score))\ntemplate = \"{n} scored {s}\"      # e.g. loaded from a config file or translation string\nprint(template.format(n=name, s=score))\n\n# f-strings — fastest, clearest, the default choice for literal templates (3.6+)\nprint(f\"{name} scored {score}\")\n",[29,1683,1684,1699,1703,1708,1733,1737,1742,1762,1785,1809,1813,1818],{"__ignoreMap":27},[32,1685,1686,1689,1691,1694,1696],{"class":34,"line":35},[32,1687,1688],{"class":64},"name, score ",[32,1690,69],{"class":68},[32,1692,1693],{"class":72}," \"Ada\"",[32,1695,1063],{"class":64},[32,1697,1698],{"class":79},"98\n",[32,1700,1701],{"class":34,"line":42},[32,1702,52],{"emptyLinePlaceholder":51},[32,1704,1705],{"class":34,"line":48},[32,1706,1707],{"class":38},"# %-formatting — legacy, still seen in logging calls and old codebases\n",[32,1709,1710,1712,1714,1716,1719,1722,1725,1727,1730],{"class":34,"line":55},[32,1711,80],{"class":79},[32,1713,83],{"class":64},[32,1715,271],{"class":72},[32,1717,1718],{"class":79},"%s",[32,1720,1721],{"class":72}," scored ",[32,1723,1724],{"class":79},"%d",[32,1726,271],{"class":72},[32,1728,1729],{"class":68}," %",[32,1731,1732],{"class":64}," (name, score))\n",[32,1734,1735],{"class":34,"line":61},[32,1736,52],{"emptyLinePlaceholder":51},[32,1738,1739],{"class":34,"line":76},[32,1740,1741],{"class":38},"# .format() — flexible, verbose, used when the template is data (not a literal)\n",[32,1743,1744,1746,1748,1750,1753,1755,1757,1759],{"class":34,"line":95},[32,1745,80],{"class":79},[32,1747,83],{"class":64},[32,1749,271],{"class":72},[32,1751,1752],{"class":79},"{}",[32,1754,1721],{"class":72},[32,1756,1752],{"class":79},[32,1758,271],{"class":72},[32,1760,1761],{"class":64},".format(name, score))\n",[32,1763,1764,1767,1769,1772,1775,1777,1780,1782],{"class":34,"line":116},[32,1765,1766],{"class":64},"template ",[32,1768,69],{"class":68},[32,1770,1771],{"class":72}," \"",[32,1773,1774],{"class":79},"{n}",[32,1776,1721],{"class":72},[32,1778,1779],{"class":79},"{s}",[32,1781,271],{"class":72},[32,1783,1784],{"class":38},"      # e.g. loaded from a config file or translation string\n",[32,1786,1787,1789,1792,1796,1798,1801,1804,1806],{"class":34,"line":122},[32,1788,80],{"class":79},[32,1790,1791],{"class":64},"(template.format(",[32,1793,1795],{"class":1794},"sCrzJ","n",[32,1797,69],{"class":68},[32,1799,1800],{"class":64},"name, ",[32,1802,1803],{"class":1794},"s",[32,1805,69],{"class":68},[32,1807,1808],{"class":64},"score))\n",[32,1810,1811],{"class":34,"line":127},[32,1812,52],{"emptyLinePlaceholder":51},[32,1814,1815],{"class":34,"line":133},[32,1816,1817],{"class":38},"# f-strings — fastest, clearest, the default choice for literal templates (3.6+)\n",[32,1819,1820,1822,1824,1826,1828,1830,1832,1834,1836,1838,1841,1843,1845],{"class":34,"line":154},[32,1821,80],{"class":79},[32,1823,83],{"class":64},[32,1825,1345],{"class":68},[32,1827,271],{"class":72},[32,1829,1350],{"class":79},[32,1831,1353],{"class":64},[32,1833,1356],{"class":79},[32,1835,1721],{"class":72},[32,1837,1350],{"class":79},[32,1839,1840],{"class":64},"score",[32,1842,1356],{"class":79},[32,1844,271],{"class":72},[32,1846,1847],{"class":64},")\n",[974,1849,1850,1853,1854,1856,1857,1859,1860,1863],{},[977,1851,1852],{},"Important distinction",": f-strings require the template to be a literal known at the point of writing the code — they can't be built from a runtime string loaded from a file or database, because the interpolation happens at parse time. ",[29,1855,1671],{}," and ",[29,1858,1675],{}," operate on ",[990,1861,1862],{},"any"," string value at runtime, which is why templating systems, translation files, and logging format strings still use them.",[14,1865,1867,1868,1870],{"id":1866},"bytes-vs-str-the-encoding-boundary","Bytes vs ",[29,1869,454],{}," — The Encoding Boundary",[974,1872,1873,1875,1876,1879,1880,1883],{},[29,1874,454],{}," is text (Unicode code points); ",[29,1877,1878],{},"bytes"," is raw binary data. Converting between them requires an explicit ",[977,1881,1882],{},"encoding",".",[19,1885,1886],{"language":21},[23,1887,1889],{"className":25,"code":1888,"language":21,"meta":27,"style":27},"text = \"café\"                    # str — a sequence of Unicode code points\nencoded = text.encode(\"utf-8\")     # bytes — b'caf\\xc3\\xa9', é takes 2 bytes in UTF-8\nprint(encoded)                       # b'caf\\xc3\\xa9'\nprint(len(text))                       # 4 — 4 code points\nprint(len(encoded))                      # 5 — 5 bytes (é is 2 bytes in UTF-8)\n\ndecoded = encoded.decode(\"utf-8\")           # back to str\nprint(decoded == text)                        # True\n\n# Wrong encoding raises or corrupts silently depending on the codec\ntry:\n    encoded.decode(\"ascii\")\nexcept UnicodeDecodeError as e:\n    print(f\"Failed: {e}\")\n",[29,1890,1891,1903,1921,1931,1945,1959,1963,1981,1995,1999,2004,2010,2020,2033],{"__ignoreMap":27},[32,1892,1893,1896,1898,1900],{"class":34,"line":35},[32,1894,1895],{"class":64},"text ",[32,1897,69],{"class":68},[32,1899,372],{"class":72},[32,1901,1902],{"class":38},"                    # str — a sequence of Unicode code points\n",[32,1904,1905,1908,1910,1913,1915,1918],{"class":34,"line":42},[32,1906,1907],{"class":64},"encoded ",[32,1909,69],{"class":68},[32,1911,1912],{"class":64}," text.encode(",[32,1914,107],{"class":72},[32,1916,1917],{"class":64},")     ",[32,1919,1920],{"class":38},"# bytes — b'caf\\xc3\\xa9', é takes 2 bytes in UTF-8\n",[32,1922,1923,1925,1928],{"class":34,"line":48},[32,1924,80],{"class":79},[32,1926,1927],{"class":64},"(encoded)                       ",[32,1929,1930],{"class":38},"# b'caf\\xc3\\xa9'\n",[32,1932,1933,1935,1937,1939,1942],{"class":34,"line":55},[32,1934,80],{"class":79},[32,1936,83],{"class":64},[32,1938,86],{"class":79},[32,1940,1941],{"class":64},"(text))                       ",[32,1943,1944],{"class":38},"# 4 — 4 code points\n",[32,1946,1947,1949,1951,1953,1956],{"class":34,"line":61},[32,1948,80],{"class":79},[32,1950,83],{"class":64},[32,1952,86],{"class":79},[32,1954,1955],{"class":64},"(encoded))                      ",[32,1957,1958],{"class":38},"# 5 — 5 bytes (é is 2 bytes in UTF-8)\n",[32,1960,1961],{"class":34,"line":76},[32,1962,52],{"emptyLinePlaceholder":51},[32,1964,1965,1968,1970,1973,1975,1978],{"class":34,"line":95},[32,1966,1967],{"class":64},"decoded ",[32,1969,69],{"class":68},[32,1971,1972],{"class":64}," encoded.decode(",[32,1974,107],{"class":72},[32,1976,1977],{"class":64},")           ",[32,1979,1980],{"class":38},"# back to str\n",[32,1982,1983,1985,1988,1990,1993],{"class":34,"line":116},[32,1984,80],{"class":79},[32,1986,1987],{"class":64},"(decoded ",[32,1989,290],{"class":68},[32,1991,1992],{"class":64}," text)                        ",[32,1994,1140],{"class":38},[32,1996,1997],{"class":34,"line":122},[32,1998,52],{"emptyLinePlaceholder":51},[32,2000,2001],{"class":34,"line":127},[32,2002,2003],{"class":38},"# Wrong encoding raises or corrupts silently depending on the codec\n",[32,2005,2006,2008],{"class":34,"line":133},[32,2007,1280],{"class":68},[32,2009,462],{"class":64},[32,2011,2012,2015,2018],{"class":34,"line":154},[32,2013,2014],{"class":64},"    encoded.decode(",[32,2016,2017],{"class":72},"\"ascii\"",[32,2019,1847],{"class":64},[32,2021,2022,2024,2027,2030],{"class":34,"line":175},[32,2023,1284],{"class":68},[32,2025,2026],{"class":79}," UnicodeDecodeError",[32,2028,2029],{"class":68}," as",[32,2031,2032],{"class":64}," e:\n",[32,2034,2035,2038,2040,2042,2045,2047,2050,2052,2054],{"class":34,"line":200},[32,2036,2037],{"class":79},"    print",[32,2039,83],{"class":64},[32,2041,1345],{"class":68},[32,2043,2044],{"class":72},"\"Failed: ",[32,2046,1350],{"class":79},[32,2048,2049],{"class":64},"e",[32,2051,1356],{"class":79},[32,2053,271],{"class":72},[32,2055,1847],{"class":64},[19,2057,2058],{"language":21},[23,2059,2061],{"className":25,"code":2060,"language":21,"meta":27,"style":27},"# Mixing str and bytes is a TypeError — Python 3 refuses to silently coerce\ntry:\n    \"hello\" + b\"world\"\nexcept TypeError as e:\n    print(e)   # can only concatenate str (not \"bytes\") to str\n",[29,2062,2063,2068,2074,2088,2099],{"__ignoreMap":27},[32,2064,2065],{"class":34,"line":35},[32,2066,2067],{"class":38},"# Mixing str and bytes is a TypeError — Python 3 refuses to silently coerce\n",[32,2069,2070,2072],{"class":34,"line":42},[32,2071,1280],{"class":68},[32,2073,462],{"class":64},[32,2075,2076,2079,2082,2085],{"class":34,"line":48},[32,2077,2078],{"class":72},"    \"hello\"",[32,2080,2081],{"class":68}," +",[32,2083,2084],{"class":68}," b",[32,2086,2087],{"class":72},"\"world\"\n",[32,2089,2090,2092,2095,2097],{"class":34,"line":55},[32,2091,1284],{"class":68},[32,2093,2094],{"class":79}," TypeError",[32,2096,2029],{"class":68},[32,2098,2032],{"class":64},[32,2100,2101,2103,2106],{"class":34,"line":61},[32,2102,2037],{"class":79},[32,2104,2105],{"class":64},"(e)   ",[32,2107,2108],{"class":38},"# can only concatenate str (not \"bytes\") to str\n",[974,2110,2111,2112,2114],{},"Python 2 blurred this line (its ",[29,2113,454],{}," was really bytes); Python 3's hard separation is one of the most important — and most disruptive to ported code — changes in the 2-to-3 migration.",[1665,2116,2118],{"id":2117},"default-encoding-pitfalls","Default encoding pitfalls",[19,2120,2121],{"language":21},[23,2122,2124],{"className":25,"code":2123,"language":21,"meta":27,"style":27},"# BAD — relies on the platform's default encoding, which varies!\n# (open() without encoding= uses locale.getpreferredencoding(), which is\n#  UTF-8 on most modern Linux\u002FmacOS but can be cp1252 on some Windows setups)\nwith open(\"data.txt\") as f:      # DANGEROUS: implicit encoding\n    content = f.read()\n\n# GOOD — always specify encoding explicitly for portable, reproducible I\u002FO\nwith open(\"data.txt\", encoding=\"utf-8\") as f:\n    content = f.read()\n",[29,2125,2126,2131,2136,2141,2165,2175,2179,2184,2209],{"__ignoreMap":27},[32,2127,2128],{"class":34,"line":35},[32,2129,2130],{"class":38},"# BAD — relies on the platform's default encoding, which varies!\n",[32,2132,2133],{"class":34,"line":42},[32,2134,2135],{"class":38},"# (open() without encoding= uses locale.getpreferredencoding(), which is\n",[32,2137,2138],{"class":34,"line":48},[32,2139,2140],{"class":38},"#  UTF-8 on most modern Linux\u002FmacOS but can be cp1252 on some Windows setups)\n",[32,2142,2143,2146,2149,2151,2154,2156,2159,2162],{"class":34,"line":55},[32,2144,2145],{"class":68},"with",[32,2147,2148],{"class":79}," open",[32,2150,83],{"class":64},[32,2152,2153],{"class":72},"\"data.txt\"",[32,2155,988],{"class":64},[32,2157,2158],{"class":68},"as",[32,2160,2161],{"class":64}," f:      ",[32,2163,2164],{"class":38},"# DANGEROUS: implicit encoding\n",[32,2166,2167,2170,2172],{"class":34,"line":61},[32,2168,2169],{"class":64},"    content ",[32,2171,69],{"class":68},[32,2173,2174],{"class":64}," f.read()\n",[32,2176,2177],{"class":34,"line":76},[32,2178,52],{"emptyLinePlaceholder":51},[32,2180,2181],{"class":34,"line":95},[32,2182,2183],{"class":38},"# GOOD — always specify encoding explicitly for portable, reproducible I\u002FO\n",[32,2185,2186,2188,2190,2192,2194,2196,2198,2200,2202,2204,2206],{"class":34,"line":116},[32,2187,2145],{"class":68},[32,2189,2148],{"class":79},[32,2191,83],{"class":64},[32,2193,2153],{"class":72},[32,2195,1063],{"class":64},[32,2197,1882],{"class":1794},[32,2199,69],{"class":68},[32,2201,107],{"class":72},[32,2203,988],{"class":64},[32,2205,2158],{"class":68},[32,2207,2208],{"class":64}," f:\n",[32,2210,2211,2213,2215],{"class":34,"line":122},[32,2212,2169],{"class":64},[32,2214,69],{"class":68},[32,2216,2174],{"class":64},[14,2218,2220,2221],{"id":2219},"regular-expressions-with-re","Regular Expressions with ",[29,2222,2223],{},"re",[19,2225,2226],{"language":21},[23,2227,2229],{"className":25,"code":2228,"language":21,"meta":27,"style":27},"import re\n\ntext = \"Contact: alice@example.com or bob@work.org\"\n\n# search — first match anywhere\nmatch = re.search(r\"[\\w.+-]+@[\\w-]+\\.[\\w.-]+\", text)\nprint(match.group())          # alice@example.com\n\n# findall — all non-overlapping matches\nemails = re.findall(r\"[\\w.+-]+@[\\w-]+\\.[\\w.-]+\", text)\nprint(emails)                   # ['alice@example.com', 'bob@work.org']\n\n# sub — replace matches\nredacted = re.sub(r\"[\\w.+-]+@[\\w-]+\\.[\\w.-]+\", \"[REDACTED]\", text)\nprint(redacted)                    # Contact: [REDACTED] or [REDACTED]\n\n# Named groups for structured extraction\npattern = re.compile(r\"(?P\u003Cuser>[\\w.+-]+)@(?P\u003Cdomain>[\\w.-]+)\")\nm = pattern.search(\"alice@example.com\")\nprint(m.group(\"user\"), m.group(\"domain\"))   # alice example.com\nprint(m.groupdict())                            # {'user': 'alice', 'domain': 'example.com'}\n",[29,2230,2231,2238,2242,2251,2255,2260,2302,2312,2316,2321,2355,2365,2369,2374,2413,2423,2427,2432,2476,2491,2512],{"__ignoreMap":27},[32,2232,2233,2235],{"class":34,"line":35},[32,2234,232],{"class":68},[32,2236,2237],{"class":64}," re\n",[32,2239,2240],{"class":34,"line":42},[32,2241,52],{"emptyLinePlaceholder":51},[32,2243,2244,2246,2248],{"class":34,"line":48},[32,2245,1895],{"class":64},[32,2247,69],{"class":68},[32,2249,2250],{"class":72}," \"Contact: alice@example.com or bob@work.org\"\n",[32,2252,2253],{"class":34,"line":55},[32,2254,52],{"emptyLinePlaceholder":51},[32,2256,2257],{"class":34,"line":61},[32,2258,2259],{"class":38},"# search — first match anywhere\n",[32,2261,2262,2265,2267,2270,2272,2274,2277,2280,2284,2287,2289,2292,2295,2297,2299],{"class":34,"line":76},[32,2263,2264],{"class":64},"match ",[32,2266,69],{"class":68},[32,2268,2269],{"class":64}," re.search(",[32,2271,752],{"class":68},[32,2273,271],{"class":72},[32,2275,2276],{"class":79},"[\\w.+-]",[32,2278,2279],{"class":68},"+",[32,2281,2283],{"class":2282},"svAP2","@",[32,2285,2286],{"class":79},"[\\w-]",[32,2288,2279],{"class":68},[32,2290,2291],{"class":757},"\\.",[32,2293,2294],{"class":79},"[\\w.-]",[32,2296,2279],{"class":68},[32,2298,271],{"class":72},[32,2300,2301],{"class":64},", text)\n",[32,2303,2304,2306,2309],{"class":34,"line":95},[32,2305,80],{"class":79},[32,2307,2308],{"class":64},"(match.group())          ",[32,2310,2311],{"class":38},"# alice@example.com\n",[32,2313,2314],{"class":34,"line":116},[32,2315,52],{"emptyLinePlaceholder":51},[32,2317,2318],{"class":34,"line":122},[32,2319,2320],{"class":38},"# findall — all non-overlapping matches\n",[32,2322,2323,2326,2328,2331,2333,2335,2337,2339,2341,2343,2345,2347,2349,2351,2353],{"class":34,"line":127},[32,2324,2325],{"class":64},"emails ",[32,2327,69],{"class":68},[32,2329,2330],{"class":64}," re.findall(",[32,2332,752],{"class":68},[32,2334,271],{"class":72},[32,2336,2276],{"class":79},[32,2338,2279],{"class":68},[32,2340,2283],{"class":2282},[32,2342,2286],{"class":79},[32,2344,2279],{"class":68},[32,2346,2291],{"class":757},[32,2348,2294],{"class":79},[32,2350,2279],{"class":68},[32,2352,271],{"class":72},[32,2354,2301],{"class":64},[32,2356,2357,2359,2362],{"class":34,"line":133},[32,2358,80],{"class":79},[32,2360,2361],{"class":64},"(emails)                   ",[32,2363,2364],{"class":38},"# ['alice@example.com', 'bob@work.org']\n",[32,2366,2367],{"class":34,"line":154},[32,2368,52],{"emptyLinePlaceholder":51},[32,2370,2371],{"class":34,"line":175},[32,2372,2373],{"class":38},"# sub — replace matches\n",[32,2375,2376,2379,2381,2384,2386,2388,2390,2392,2394,2396,2398,2400,2402,2404,2406,2408,2411],{"class":34,"line":200},[32,2377,2378],{"class":64},"redacted ",[32,2380,69],{"class":68},[32,2382,2383],{"class":64}," re.sub(",[32,2385,752],{"class":68},[32,2387,271],{"class":72},[32,2389,2276],{"class":79},[32,2391,2279],{"class":68},[32,2393,2283],{"class":2282},[32,2395,2286],{"class":79},[32,2397,2279],{"class":68},[32,2399,2291],{"class":757},[32,2401,2294],{"class":79},[32,2403,2279],{"class":68},[32,2405,271],{"class":72},[32,2407,1063],{"class":64},[32,2409,2410],{"class":72},"\"[REDACTED]\"",[32,2412,2301],{"class":64},[32,2414,2415,2417,2420],{"class":34,"line":205},[32,2416,80],{"class":79},[32,2418,2419],{"class":64},"(redacted)                    ",[32,2421,2422],{"class":38},"# Contact: [REDACTED] or [REDACTED]\n",[32,2424,2425],{"class":34,"line":211},[32,2426,52],{"emptyLinePlaceholder":51},[32,2428,2429],{"class":34,"line":217},[32,2430,2431],{"class":38},"# Named groups for structured extraction\n",[32,2433,2434,2437,2439,2442,2444,2446,2448,2452,2454,2456,2459,2461,2463,2466,2468,2470,2472,2474],{"class":34,"line":223},[32,2435,2436],{"class":64},"pattern ",[32,2438,69],{"class":68},[32,2440,2441],{"class":64}," re.compile(",[32,2443,752],{"class":68},[32,2445,271],{"class":72},[32,2447,83],{"class":79},[32,2449,2451],{"class":2450},"sk71V","?P\u003Cuser>",[32,2453,2276],{"class":79},[32,2455,2279],{"class":68},[32,2457,2458],{"class":79},")",[32,2460,2283],{"class":2282},[32,2462,83],{"class":79},[32,2464,2465],{"class":2450},"?P\u003Cdomain>",[32,2467,2294],{"class":79},[32,2469,2279],{"class":68},[32,2471,2458],{"class":79},[32,2473,271],{"class":72},[32,2475,1847],{"class":64},[32,2477,2478,2481,2483,2486,2489],{"class":34,"line":229},[32,2479,2480],{"class":64},"m ",[32,2482,69],{"class":68},[32,2484,2485],{"class":64}," pattern.search(",[32,2487,2488],{"class":72},"\"alice@example.com\"",[32,2490,1847],{"class":64},[32,2492,2493,2495,2498,2501,2504,2507,2509],{"class":34,"line":238},[32,2494,80],{"class":79},[32,2496,2497],{"class":64},"(m.group(",[32,2499,2500],{"class":72},"\"user\"",[32,2502,2503],{"class":64},"), m.group(",[32,2505,2506],{"class":72},"\"domain\"",[32,2508,808],{"class":64},[32,2510,2511],{"class":38},"# alice example.com\n",[32,2513,2514,2516,2519],{"class":34,"line":243},[32,2515,80],{"class":79},[32,2517,2518],{"class":64},"(m.groupdict())                            ",[32,2520,2521],{"class":38},"# {'user': 'alice', 'domain': 'example.com'}\n",[1665,2523,2525],{"id":2524},"compiling-patterns-for-reuse-a-real-performance-consideration","Compiling patterns for reuse — a real performance consideration",[19,2527,2528],{"language":21},[23,2529,2531],{"className":25,"code":2530,"language":21,"meta":27,"style":27},"# Compile once, reuse many times, in hot paths (loops, per-request validation)\nEMAIL_RE = re.compile(r\"^[\\w.+-]+@[\\w-]+\\.[\\w.-]+$\")\n\ndef is_valid_email(s: str) -> bool:\n    return bool(EMAIL_RE.match(s))    # match() anchors at the START only, not the end!\n\nprint(is_valid_email(\"alice@example.com\"))     # True\nprint(is_valid_email(\"alice@example.com\\nmalicious\"))   # True in re.match with $!\n",[29,2532,2533,2538,2576,2580,2598,2615,2619,2633],{"__ignoreMap":27},[32,2534,2535],{"class":34,"line":35},[32,2536,2537],{"class":38},"# Compile once, reuse many times, in hot paths (loops, per-request validation)\n",[32,2539,2540,2543,2546,2548,2550,2552,2555,2557,2559,2561,2563,2565,2567,2569,2572,2574],{"class":34,"line":42},[32,2541,2542],{"class":79},"EMAIL_RE",[32,2544,2545],{"class":68}," =",[32,2547,2441],{"class":64},[32,2549,752],{"class":68},[32,2551,271],{"class":72},[32,2553,2554],{"class":79},"^[\\w.+-]",[32,2556,2279],{"class":68},[32,2558,2283],{"class":2282},[32,2560,2286],{"class":79},[32,2562,2279],{"class":68},[32,2564,2291],{"class":757},[32,2566,2294],{"class":79},[32,2568,2279],{"class":68},[32,2570,2571],{"class":79},"$",[32,2573,271],{"class":72},[32,2575,1847],{"class":64},[32,2577,2578],{"class":34,"line":48},[32,2579,52],{"emptyLinePlaceholder":51},[32,2581,2582,2584,2587,2589,2591,2593,2596],{"class":34,"line":55},[32,2583,444],{"class":68},[32,2585,2586],{"class":447}," is_valid_email",[32,2588,451],{"class":64},[32,2590,454],{"class":79},[32,2592,457],{"class":64},[32,2594,2595],{"class":79},"bool",[32,2597,462],{"class":64},[32,2599,2600,2602,2605,2607,2609,2612],{"class":34,"line":61},[32,2601,468],{"class":68},[32,2603,2604],{"class":79}," bool",[32,2606,83],{"class":64},[32,2608,2542],{"class":79},[32,2610,2611],{"class":64},".match(s))    ",[32,2613,2614],{"class":38},"# match() anchors at the START only, not the end!\n",[32,2616,2617],{"class":34,"line":76},[32,2618,52],{"emptyLinePlaceholder":51},[32,2620,2621,2623,2626,2628,2631],{"class":34,"line":95},[32,2622,80],{"class":79},[32,2624,2625],{"class":64},"(is_valid_email(",[32,2627,2488],{"class":72},[32,2629,2630],{"class":64},"))     ",[32,2632,1140],{"class":38},[32,2634,2635,2637,2639,2642,2645,2648,2650],{"class":34,"line":116},[32,2636,80],{"class":79},[32,2638,2625],{"class":64},[32,2640,2641],{"class":72},"\"alice@example.com",[32,2643,2644],{"class":79},"\\n",[32,2646,2647],{"class":72},"malicious\"",[32,2649,808],{"class":64},[32,2651,2652],{"class":38},"# True in re.match with $!\n",[974,2654,2655,2658,2659,2661,2662,2665,2666,2669,2670,2673,2674,2676,2677,2679,2680,2683,2684,2686,2687,1883],{},[977,2656,2657],{},"Gotcha",": ",[29,2660,2571],{}," in a regex matches the end of the string ",[977,2663,2664],{},"or just before a trailing newline"," — it is not a strict end-of-string anchor. ",[29,2667,2668],{},"\"alice@example.com\\nmalicious\""," matches ",[29,2671,2672],{},"^[\\w.+-]+@[\\w-]+\\.[\\w.-]+$"," because ",[29,2675,2571],{}," allows a trailing ",[29,2678,2644],{}," before the true end. For strict whole-string validation (security-sensitive contexts especially — see chapter 27), use ",[29,2681,2682],{},"\\Z"," instead of ",[29,2685,2571],{},", or ",[29,2688,2689],{},"re.fullmatch()",[19,2691,2692],{"language":21},[23,2693,2695],{"className":25,"code":2694,"language":21,"meta":27,"style":27},"STRICT_EMAIL_RE = re.compile(r\"^[\\w.+-]+@[\\w-]+\\.[\\w.-]+\\Z\")\n\ndef is_valid_email_strict(s: str) -> bool:\n    return bool(STRICT_EMAIL_RE.match(s))\n\nprint(is_valid_email_strict(\"alice@example.com\\nmalicious\"))   # False — correctly rejected\n\n# Or, more idiomatically:\ndef is_valid_email_fullmatch(s: str) -> bool:\n    return bool(re.fullmatch(r\"[\\w.+-]+@[\\w-]+\\.[\\w.-]+\", s))\n\nprint(is_valid_email_fullmatch(\"alice@example.com\\nmalicious\"))   # False\n",[29,2696,2697,2732,2736,2753,2766,2770,2788,2792,2797,2814,2848,2852],{"__ignoreMap":27},[32,2698,2699,2702,2704,2706,2708,2710,2712,2714,2716,2718,2720,2722,2724,2726,2728,2730],{"class":34,"line":35},[32,2700,2701],{"class":79},"STRICT_EMAIL_RE",[32,2703,2545],{"class":68},[32,2705,2441],{"class":64},[32,2707,752],{"class":68},[32,2709,271],{"class":72},[32,2711,2554],{"class":79},[32,2713,2279],{"class":68},[32,2715,2283],{"class":2282},[32,2717,2286],{"class":79},[32,2719,2279],{"class":68},[32,2721,2291],{"class":757},[32,2723,2294],{"class":79},[32,2725,2279],{"class":68},[32,2727,2682],{"class":79},[32,2729,271],{"class":72},[32,2731,1847],{"class":64},[32,2733,2734],{"class":34,"line":42},[32,2735,52],{"emptyLinePlaceholder":51},[32,2737,2738,2740,2743,2745,2747,2749,2751],{"class":34,"line":48},[32,2739,444],{"class":68},[32,2741,2742],{"class":447}," is_valid_email_strict",[32,2744,451],{"class":64},[32,2746,454],{"class":79},[32,2748,457],{"class":64},[32,2750,2595],{"class":79},[32,2752,462],{"class":64},[32,2754,2755,2757,2759,2761,2763],{"class":34,"line":55},[32,2756,468],{"class":68},[32,2758,2604],{"class":79},[32,2760,83],{"class":64},[32,2762,2701],{"class":79},[32,2764,2765],{"class":64},".match(s))\n",[32,2767,2768],{"class":34,"line":61},[32,2769,52],{"emptyLinePlaceholder":51},[32,2771,2772,2774,2777,2779,2781,2783,2785],{"class":34,"line":76},[32,2773,80],{"class":79},[32,2775,2776],{"class":64},"(is_valid_email_strict(",[32,2778,2641],{"class":72},[32,2780,2644],{"class":79},[32,2782,2647],{"class":72},[32,2784,808],{"class":64},[32,2786,2787],{"class":38},"# False — correctly rejected\n",[32,2789,2790],{"class":34,"line":95},[32,2791,52],{"emptyLinePlaceholder":51},[32,2793,2794],{"class":34,"line":116},[32,2795,2796],{"class":38},"# Or, more idiomatically:\n",[32,2798,2799,2801,2804,2806,2808,2810,2812],{"class":34,"line":122},[32,2800,444],{"class":68},[32,2802,2803],{"class":447}," is_valid_email_fullmatch",[32,2805,451],{"class":64},[32,2807,454],{"class":79},[32,2809,457],{"class":64},[32,2811,2595],{"class":79},[32,2813,462],{"class":64},[32,2815,2816,2818,2820,2823,2825,2827,2829,2831,2833,2835,2837,2839,2841,2843,2845],{"class":34,"line":127},[32,2817,468],{"class":68},[32,2819,2604],{"class":79},[32,2821,2822],{"class":64},"(re.fullmatch(",[32,2824,752],{"class":68},[32,2826,271],{"class":72},[32,2828,2276],{"class":79},[32,2830,2279],{"class":68},[32,2832,2283],{"class":2282},[32,2834,2286],{"class":79},[32,2836,2279],{"class":68},[32,2838,2291],{"class":757},[32,2840,2294],{"class":79},[32,2842,2279],{"class":68},[32,2844,271],{"class":72},[32,2846,2847],{"class":64},", s))\n",[32,2849,2850],{"class":34,"line":133},[32,2851,52],{"emptyLinePlaceholder":51},[32,2853,2854,2856,2859,2861,2863,2865,2867],{"class":34,"line":154},[32,2855,80],{"class":79},[32,2857,2858],{"class":64},"(is_valid_email_fullmatch(",[32,2860,2641],{"class":72},[32,2862,2644],{"class":79},[32,2864,2647],{"class":72},[32,2866,808],{"class":64},[32,2868,2869],{"class":38},"# False\n",[14,2871,2873],{"id":2872},"string-building-concatenation-performance","String Building — Concatenation Performance",[19,2875,2876],{"language":21},[23,2877,2879],{"className":25,"code":2878,"language":21,"meta":27,"style":27},"# BAD in a loop — each += creates a brand-new string (strings are immutable),\n# making naive concatenation O(n^2) for n appends in the worst case\nresult = \"\"\nfor i in range(10000):\n    result += str(i)   # allocates a new string of growing size EVERY iteration\n\n# GOOD — collect pieces, join once (join is O(n))\nparts = [str(i) for i in range(10000)]\nresult = \"\".join(parts)\n",[29,2880,2881,2886,2891,2901,2921,2938,2942,2947,2976],{"__ignoreMap":27},[32,2882,2883],{"class":34,"line":35},[32,2884,2885],{"class":38},"# BAD in a loop — each += creates a brand-new string (strings are immutable),\n",[32,2887,2888],{"class":34,"line":42},[32,2889,2890],{"class":38},"# making naive concatenation O(n^2) for n appends in the worst case\n",[32,2892,2893,2896,2898],{"class":34,"line":48},[32,2894,2895],{"class":64},"result ",[32,2897,69],{"class":68},[32,2899,2900],{"class":72}," \"\"\n",[32,2902,2903,2905,2908,2910,2913,2915,2918],{"class":34,"line":55},[32,2904,578],{"class":68},[32,2906,2907],{"class":64}," i ",[32,2909,421],{"class":68},[32,2911,2912],{"class":79}," range",[32,2914,83],{"class":64},[32,2916,2917],{"class":79},"10000",[32,2919,2920],{"class":64},"):\n",[32,2922,2923,2926,2929,2932,2935],{"class":34,"line":61},[32,2924,2925],{"class":64},"    result ",[32,2927,2928],{"class":68},"+=",[32,2930,2931],{"class":79}," str",[32,2933,2934],{"class":64},"(i)   ",[32,2936,2937],{"class":38},"# allocates a new string of growing size EVERY iteration\n",[32,2939,2940],{"class":34,"line":76},[32,2941,52],{"emptyLinePlaceholder":51},[32,2943,2944],{"class":34,"line":95},[32,2945,2946],{"class":38},"# GOOD — collect pieces, join once (join is O(n))\n",[32,2948,2949,2952,2954,2956,2958,2961,2963,2965,2967,2969,2971,2973],{"class":34,"line":116},[32,2950,2951],{"class":64},"parts ",[32,2953,69],{"class":68},[32,2955,1623],{"class":64},[32,2957,454],{"class":79},[32,2959,2960],{"class":64},"(i) ",[32,2962,578],{"class":68},[32,2964,2907],{"class":64},[32,2966,421],{"class":68},[32,2968,2912],{"class":79},[32,2970,83],{"class":64},[32,2972,2917],{"class":79},[32,2974,2975],{"class":64},")]\n",[32,2977,2978,2980,2982,2984],{"class":34,"line":122},[32,2979,2895],{"class":64},[32,2981,69],{"class":68},[32,2983,769],{"class":72},[32,2985,2986],{"class":64},".join(parts)\n",[974,2988,2989,2990,2992,2993,2996],{},"CPython has an optimization (",[29,2991,454],{}," concatenation in a loop can sometimes resize in place when the left operand has refcount 1) that mitigates this in some cases, but it is a CPython implementation detail, not a language guarantee — ",[29,2994,2995],{},"\"\".join(...)"," is the portable, always-correct-and-fast idiom, and is what experienced Python developers reach for reflexively.",[14,2998,3000],{"id":2999},"tips-tricks","💡 Tips & Tricks",[3002,3003,3004,3025,3038,3054,3066],"ul",{},[3005,3006,3007,3016,3017,3020,3021,3024],"li",{},[977,3008,3009,3012,3013,3015],{},[29,3010,3011],{},"str.format_map"," and f-strings with ",[29,3014,69],{}," for debugging"," — ",[29,3018,3019],{},"f\"{some_expr=}\""," prints both the source expression text and its value, replacing manual ",[29,3022,3023],{},"print(\"some_expr:\", some_expr)"," debug lines.",[3005,3026,3027,3033,3034,3037],{},[977,3028,3029,3032],{},[29,3030,3031],{},"textwrap.dedent"," cleans up triple-quoted strings indented to match code"," — writing a multi-line string inside an indented function body normally bakes in the indentation; ",[29,3035,3036],{},"textwrap.dedent(s)"," strips the common leading whitespace.",[3005,3039,3040,3049,3050,3053],{},[977,3041,3042,1277,3045,3048],{},[29,3043,3044],{},"str.translate",[29,3046,3047],{},"str.maketrans"," for fast bulk character replacement"," — replacing many individual characters is far faster via a translation table than chained ",[29,3051,3052],{},".replace()"," calls.",[3005,3055,3056,3016,3062,3065],{},[977,3057,3058,3061],{},[29,3059,3060],{},"re.VERBOSE"," for readable complex patterns",[29,3063,3064],{},"re.compile(r\"\"\"\\d{3} -\\d{4}\"\"\", re.VERBOSE)"," lets you add whitespace and comments inside a pattern for maintainability; whitespace in the pattern is ignored unless escaped or in a character class.",[3005,3067,3068,3074,3075,3077,3078,3081,3082,3085],{},[977,3069,3070,3073],{},[29,3071,3072],{},"unicodedata.normalize"," before comparing user-supplied Unicode text"," — visually identical strings can have different underlying code point sequences (",[29,3076,191],{}," as one composed code point vs. ",[29,3079,3080],{},"\"e\" + combining acute accent","); normalize with ",[29,3083,3084],{},"NFC"," before equality checks or lookups.",[14,3087,3089],{"id":3088},"️-edge-cases-gotchas","⚠️ Edge Cases & Gotchas",[3002,3091,3092,3117,3140,3162,3188],{},[3005,3093,3094,3100,3101,1359,3104,3106,3107,1359,3110,3112,3113,3116],{},[977,3095,3096,3099],{},[29,3097,3098],{},"len()"," counts code points, not \"characters\" as a human perceives them, and not bytes"," — a single visually-perceived emoji or accented character can be one code point, multiple combining code points, or (in UTF-8 bytes) up to 4 bytes — ",[29,3102,3103],{},"len(\"café\")",[29,3105,805],{}," but ",[29,3108,3109],{},"len(\"café\".encode())",[29,3111,900],{},", and something like a flag emoji built from regional indicator pairs can have ",[29,3114,3115],{},"len() == 2"," for what looks like one glyph.",[3005,3118,3119,3132,3133,3135,3136,3139],{},[977,3120,3121,3124,3125,3128,3129],{},[29,3122,3123],{},"str.format()","\u002Ff-strings silently call ",[29,3126,3127],{},"__format__",", which can differ wildly from ",[29,3130,3131],{},"__str__"," — custom objects that define ",[29,3134,3127],{}," (e.g., for locale-aware number formatting) can produce output in an f-string that doesn't match ",[29,3137,3138],{},"print(obj)","; this is rare but confusing when it happens.",[3005,3141,3142,3151,3152,3155,3156,3158,3159,3161],{},[977,3143,3144,3145,3147,3148,3150],{},"Regex ",[29,3146,2571],{}," matches before a trailing ",[29,3149,2644],{},", not strictly \"end of string\""," — as shown above, naive ",[29,3153,3154],{},"^...$"," \"validation\" patterns can be bypassed by appending a newline plus arbitrary content; use ",[29,3157,2682],{}," or ",[29,3160,2689],{}," for security-sensitive validation.",[3005,3163,3164,3167,3168,3171,3172,3175,3176,3179,3180,3183,3184,3187],{},[977,3165,3166],{},"Implicit default encoding varies by platform"," — omitting ",[29,3169,3170],{},"encoding="," in ",[29,3173,3174],{},"open()"," relies on ",[29,3177,3178],{},"locale.getpreferredencoding()",", which is usually UTF-8 on modern Linux\u002FmacOS but historically defaulted to something like ",[29,3181,3182],{},"cp1252"," on Windows — always pass ",[29,3185,3186],{},"encoding=\"utf-8\""," explicitly for reproducible cross-platform behavior.",[3005,3189,3190,3196,3197,3199],{},[977,3191,3192,3193,3195],{},"String concatenation with ",[29,3194,2928],{}," in a loop is O(n²) in the general case, despite sometimes appearing fast in CPython due to an internal optimization"," — that optimization only triggers when the string has a reference count of exactly 1 and is CPython-specific (not guaranteed by the language, and absent in PyPy\u002FJython) — always use ",[29,3198,2995],{}," for building strings from many pieces in production code.",[14,3201,3203],{"id":3202},"spot-the-bug","🧠 Spot the Bug",[974,3205,3206],{},"What does this print, and why does the count look wrong to a beginner?",[19,3208,3209],{"language":21},[23,3210,3212],{"className":25,"code":3211,"language":21,"meta":27,"style":27},"flag = \"🇺🇸\"\nname = \"café\"\n\nprint(len(flag))\nprint(len(name))\nprint(name[3])\nprint(name.encode(\"utf-8\")[3])\n",[29,3213,3214,3223,3231,3235,3246,3257,3270],{"__ignoreMap":27},[32,3215,3216,3218,3220],{"class":34,"line":35},[32,3217,537],{"class":64},[32,3219,69],{"class":68},[32,3221,3222],{"class":72}," \"🇺🇸\"\n",[32,3224,3225,3227,3229],{"class":34,"line":42},[32,3226,1307],{"class":64},[32,3228,69],{"class":68},[32,3230,73],{"class":72},[32,3232,3233],{"class":34,"line":48},[32,3234,52],{"emptyLinePlaceholder":51},[32,3236,3237,3239,3241,3243],{"class":34,"line":55},[32,3238,80],{"class":79},[32,3240,83],{"class":64},[32,3242,86],{"class":79},[32,3244,3245],{"class":64},"(flag))\n",[32,3247,3248,3250,3252,3254],{"class":34,"line":61},[32,3249,80],{"class":79},[32,3251,83],{"class":64},[32,3253,86],{"class":79},[32,3255,3256],{"class":64},"(name))\n",[32,3258,3259,3261,3264,3267],{"class":34,"line":76},[32,3260,80],{"class":79},[32,3262,3263],{"class":64},"(name[",[32,3265,3266],{"class":79},"3",[32,3268,3269],{"class":64},"])\n",[32,3271,3272,3274,3277,3279,3282,3284],{"class":34,"line":95},[32,3273,80],{"class":79},[32,3275,3276],{"class":64},"(name.encode(",[32,3278,107],{"class":72},[32,3280,3281],{"class":64},")[",[32,3283,3266],{"class":79},[32,3285,3269],{"class":64},[3287,3288,3289,3293,3353],"details",{},[3290,3291,3292],"summary",{},"Answer",[974,3294,3295,1359,3298,3300,3301,1359,3304,3016,3306,3309,3310,1063,3313,1063,3316,1063,3318,3321,3322,1359,3325,3328,3329,1359,3332,3335,3336,3338,3339,3341,3342,3344,3345,3348,3349,3352],{},[29,3296,3297],{},"len(flag)",[29,3299,946],{}," — the US flag emoji is not one code point; it's two \"regional indicator symbol\" code points (🇺 + 🇸) that renderers combine visually into a single flag glyph. ",[29,3302,3303],{},"len(name)",[29,3305,805],{},[29,3307,3308],{},"\"café\""," has 4 Unicode code points (",[29,3311,3312],{},"c",[29,3314,3315],{},"a",[29,3317,1345],{},[29,3319,3320],{},"é","), each counted as one, regardless of how many bytes they take when encoded. ",[29,3323,3324],{},"name[3]",[29,3326,3327],{},"'é'"," — indexing operates on code points, giving the whole accented character as one unit. ",[29,3330,3331],{},"name.encode(\"utf-8\")[3]",[29,3333,3334],{},"169"," — an integer, because indexing a ",[29,3337,1878],{}," object returns an ",[29,3340,680],{}," (the byte value), not a length-1 ",[29,3343,1878],{}," object, and it's the ",[990,3346,3347],{},"second"," byte of é's 2-byte UTF-8 encoding (",[29,3350,3351],{},"0xc3 0xa9"," = 195, 169), not the character itself.",[974,3354,3355,2658,3358,3360,3361,3363],{},[977,3356,3357],{},"The lesson",[29,3359,3098],{}," and indexing on ",[29,3362,454],{}," operate on Unicode code points, which don't correspond 1:1 with either \"user-perceived characters\" (grapheme clusters, which may span multiple code points) or bytes (which vary 1–4 per code point in UTF-8) — never assume any of the three counts match.",[14,3365,3367],{"id":3366},"key-takeaways","Key Takeaways",[3002,3369,3370,3373,3381,3396,3415],{},[3005,3371,3372],{},"Strings are immutable sequences of Unicode code points; every \"modification\" method returns a new string.",[3005,3374,3375,3376,1281,3378,3380],{},"f-strings are the preferred formatting mechanism for literal templates known at write-time; ",[29,3377,1671],{},[29,3379,1675],{}," remain necessary for runtime-supplied templates (config files, i18n).",[3005,3382,3383,3385,3386,3388,3389,1281,3392,3395],{},[29,3384,454],{}," (text) and ",[29,3387,1878],{}," (binary data) are strictly separate types in Python 3 — conversion requires an explicit ",[29,3390,3391],{},".encode()",[29,3393,3394],{},".decode()"," with a named codec; never rely on the platform default encoding.",[3005,3397,3398,3401,3402,1281,3404,2683,3406,1281,3408,3411,3412,3414],{},[29,3399,3400],{},"re.compile()"," your patterns once and reuse them in hot paths; use ",[29,3403,2682],{},[29,3405,2689],{},[29,3407,2571],{},[29,3409,3410],{},"re.match()"," when validating untrusted input, since ",[29,3413,2571],{}," tolerates a trailing newline.",[3005,3416,3417,3418,3421,3422,3424],{},"Build strings from many pieces with ",[29,3419,3420],{},"\"\".join(parts)",", not repeated ",[29,3423,2928],{}," in a loop — the latter is O(n²) in the general case and relies on a non-portable CPython optimization to ever be fast.",[3426,3427,3428],"style",{},"html pre.shiki code .sdCPZ, html code.shiki .sdCPZ{--shiki-default:#6A737D;--shiki-github-dark:#6A737D}html pre.shiki code .ssxIu, html code.shiki .ssxIu{--shiki-default:#24292E;--shiki-github-dark:#E1E4E8}html pre.shiki code .svdQ7, html code.shiki .svdQ7{--shiki-default:#D73A49;--shiki-github-dark:#F97583}html pre.shiki code .sJ6F3, html code.shiki .sJ6F3{--shiki-default:#032F62;--shiki-github-dark:#9ECBFF}html pre.shiki code .snvgF, html code.shiki .snvgF{--shiki-default:#005CC5;--shiki-github-dark:#79B8FF}html pre.shiki code .sIsaT, html code.shiki .sIsaT{--shiki-default:#6F42C1;--shiki-github-dark:#B392F0}html .default .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .github-dark .shiki span {color: var(--shiki-github-dark);background: var(--shiki-github-dark-bg);font-style: var(--shiki-github-dark-font-style);font-weight: var(--shiki-github-dark-font-weight);text-decoration: var(--shiki-github-dark-text-decoration);}html.github-dark .shiki span {color: var(--shiki-github-dark);background: var(--shiki-github-dark-bg);font-style: var(--shiki-github-dark-font-style);font-weight: var(--shiki-github-dark-font-weight);text-decoration: var(--shiki-github-dark-text-decoration);}html pre.shiki code .snRuI, html code.shiki .snRuI{--shiki-default:#22863A;--shiki-default-font-weight:bold;--shiki-github-dark:#85E89D;--shiki-github-dark-font-weight:bold}html pre.shiki code .sCrzJ, html code.shiki .sCrzJ{--shiki-default:#E36209;--shiki-github-dark:#FFAB70}html pre.shiki code .svAP2, html code.shiki .svAP2{--shiki-default:#032F62;--shiki-github-dark:#DBEDFF}html pre.shiki code .sk71V, html code.shiki .sk71V{--shiki-default:#22863A;--shiki-github-dark:#85E89D}",{"title":27,"searchDepth":42,"depth":42,"links":3430},[3431,3432,3433,3434,3438,3442,3446,3447,3448,3449,3450],{"id":16,"depth":42,"text":17},{"id":819,"depth":42,"text":820},{"id":999,"depth":42,"text":1000},{"id":1294,"depth":42,"text":1295,"children":3435},[3436],{"id":1667,"depth":48,"text":3437},"f-strings vs .format() vs % — know all three, prefer f-strings",{"id":1866,"depth":42,"text":3439,"children":3440},"Bytes vs str — The Encoding Boundary",[3441],{"id":2117,"depth":48,"text":2118},{"id":2219,"depth":42,"text":3443,"children":3444},"Regular Expressions with re",[3445],{"id":2524,"depth":48,"text":2525},{"id":2872,"depth":42,"text":2873},{"id":2999,"depth":42,"text":3000},{"id":3088,"depth":42,"text":3089},{"id":3202,"depth":42,"text":3203},{"id":3366,"depth":42,"text":3367},"md",{},"\u002Fpython\u002F06-strings-and-text",{"title":5,"description":27},"python\u002F06-strings-and-text","TkWC5ioT6sGOh3vNhiPkuk-MM3XgsxPfJ9TNDuOSD4g",1789924651341]