[{"data":1,"prerenderedAt":2318},["ShallowReactive",2],{"doc:\u002Fadvanced-data-transformation-and-cleaning\u002Fexcel-formula-equivalents-in-pandas\u002Findex-match-equivalent-in-pandas":3,"surround:\u002Fadvanced-data-transformation-and-cleaning\u002Fexcel-formula-equivalents-in-pandas\u002Findex-match-equivalent-in-pandas":2309},{"id":4,"title":5,"body":6,"dateModified":2279,"datePublished":2279,"description":2280,"extension":2281,"faq":2282,"meta":2293,"navigation":219,"path":2302,"seo":2303,"slug":2305,"stem":2306,"type":2307,"__hash__":2308},"docs\u002Fadvanced-data-transformation-and-cleaning\u002Fexcel-formula-equivalents-in-pandas\u002Findex-match-equivalent-in-pandas\u002Findex.md","INDEX MATCH Equivalent in pandas",{"type":7,"value":8,"toc":2264},"minimark",[9,32,159,164,191,479,483,489,581,594,599,662,666,672,765,780,784,787,878,881,910,914,988,994,1148,1159,1170,1246,1250,1256,1411,1425,1429,1432,1653,1664,1668,1674,1842,1845,1849,1963,1967,2048,2051,2149,2156,2162,2166,2181,2185,2192,2198,2204,2214,2220,2224,2260],[10,11,12,13,17,18,21,22,27,28,31],"p",{},"INDEX\u002FMATCH exists because VLOOKUP cannot look leftwards. pandas has no such limitation, so the pair\ncollapses into either ",[14,15,16],"code",{},"merge"," or ",[14,19,20],{},"map"," depending on how much you need back — and the awkward part of\nthe Excel version, keeping the two ranges aligned, simply disappears. This guide, part of\n",[23,24,26],"a",{"href":25},"\u002Fadvanced-data-transformation-and-cleaning\u002Fexcel-formula-equivalents-in-pandas\u002F","Excel Formula Equivalents in pandas",",\ncovers both forms, composite keys, and the approximate-match case that ",[14,29,30],{},"merge_asof"," handles.",[33,34,42,43,42,47,42,51,42,58,42,68,42,74,42,81,42,86,42,90,42,94,42,98,42,103,42,107,42,111,42,114,42,117,42,120,42,123,42,131,42,137,42,142,42,147,42,151,42,154],"svg",{"viewBox":35,"role":36,"ariaLabelledBy":37,"xmlns":40,"style":41},"0 0 760 224","img",[38,39],"im-two-t","im-two-d","http:\u002F\u002Fwww.w3.org\u002F2000\u002Fsvg","width:100%;max-width:760px;height:auto;display:block;margin:1.5rem auto;font-family:Inter,ui-sans-serif,system-ui,sans-serif","\n  ",[44,45,46],"title",{"id":38},"One column back, or several",[48,49,50],"desc",{"id":39},"Mapping from a Series keyed by the lookup column adds exactly one column and cannot change the row count, while merge brings several columns at once and can multiply rows if the key repeats.",[52,53],"rect",{"x":54,"y":54,"width":55,"height":56,"fill":57},"0","760","224","#ffffff",[52,59],{"x":60,"y":61,"width":62,"height":63,"rx":64,"fill":65,"stroke":66,"style":67},"20","28","270.0","162","14","#d9f4f1","var(--teal,#0f9488)","stroke-width:2px",[69,70,20],"text",{"x":71,"y":72,"style":73},"155.0","54","font-size:13px;font-weight:700;fill:var(--teal-ink,#0b6157);text-anchor:middle",[75,76],"line",{"x1":77,"y1":78,"x2":79,"y2":78,"stroke":66,"style":80},"36","64","274.0","stroke-width:1px",[69,82,85],{"x":71,"y":83,"style":84},"86","font-size:11.5px;font-weight:400;fill:var(--text,#172033);text-anchor:middle","one column back",[69,87,89],{"x":71,"y":88,"style":84},"109","row count is safe",[69,91,93],{"x":71,"y":92,"style":84},"132","takes a dict too",[69,95,97],{"x":71,"y":96,"style":84},"155","shortest form",[52,99],{"x":100,"y":61,"width":62,"height":63,"rx":64,"fill":101,"stroke":102,"style":67},"470.0","#f0f4ff","var(--brand,#5b5cf0)",[69,104,16],{"x":105,"y":72,"style":106},"605.0","font-size:13px;font-weight:700;fill:var(--brand-strong,#4338ca);text-anchor:middle",[75,108],{"x1":109,"y1":78,"x2":110,"y2":78,"stroke":102,"style":80},"486.0","724.0",[69,112,113],{"x":105,"y":83,"style":84},"many columns at once",[69,115,116],{"x":105,"y":88,"style":84},"composite keys",[69,118,119],{"x":105,"y":92,"style":84},"join type is explicit",[69,121,122],{"x":105,"y":96,"style":84},"validate the key",[52,124],{"x":125,"y":126,"width":127,"height":128,"rx":129,"fill":130,"stroke":102},"316.0","90.0","128","38","19","#ebebfd",[69,132,136],{"x":133,"y":134,"style":135},"380.0","114.0","font-size:12.5px;font-weight:700;fill:var(--brand-strong,#4338ca);text-anchor:middle","choose by",[75,138],{"x1":139,"y1":140,"x2":141,"y2":140,"stroke":102,"style":67},"295.0","109.0","309.0",[143,144],"polygon",{"points":145,"fill":146},"309.0,109.0 300.0,104.0 300.0,114.0","#5b5cf0",[75,148],{"x1":149,"y1":140,"x2":150,"y2":140,"stroke":102,"style":67},"449.0","463.0",[143,152],{"points":153,"fill":146},"463.0,109.0 454.0,104.0 454.0,114.0",[69,155,158],{"x":133,"y":156,"style":157},"210","font-size:12.5px;font-weight:400;fill:var(--muted,#5b6780);text-anchor:middle","how much you need back decides the tool",[160,161,163],"h2",{"id":162},"prerequisites","Prerequisites",[165,166,171],"pre",{"className":167,"code":168,"language":169,"meta":170,"style":170},"language-bash shiki shiki-themes github-light github-dark-high-contrast","pip install pandas openpyxl\n","bash","",[14,172,173],{"__ignoreMap":170},[174,175,177,181,185,188],"span",{"class":75,"line":176},1,[174,178,180],{"class":179},"sMTad","pip",[174,182,184],{"class":183},"srMev"," install",[174,186,187],{"class":183}," pandas",[174,189,190],{"class":183}," openpyxl\n",[165,192,196],{"className":193,"code":194,"language":195,"meta":170,"style":170},"language-python shiki shiki-themes github-light github-dark-high-contrast","import pandas as pd\n\norders = pd.DataFrame({\n    \"Order\": [1001, 1002, 1003, 1004, 1005],\n    \"SKU\": [\"A-100\", \"B-200\", \"A-100\", \"C-300\", \"D-400\"],\n    \"Region\": [\"North\", \"South\", \"West\", \"North\", \"South\"],\n    \"Units\": [12, 4, 9, 22, 7],\n})\n\nproducts = pd.DataFrame({\n    \"SKU\": [\"A-100\", \"B-200\", \"C-300\"],\n    \"Description\": [\"Widget, small\", \"Gadget\", \"Widget, large\"],\n    \"Unit_Price\": [19.99, 49.50, 34.75],\n    \"Category\": [\"Widgets\", \"Gadgets\", \"Widgets\"],\n})\n","python",[14,197,198,214,221,233,270,302,333,366,372,377,387,406,429,452,474],{"__ignoreMap":170},[174,199,200,204,208,211],{"class":75,"line":176},[174,201,203],{"class":202},"s-kum","import",[174,205,207],{"class":206},"skGVy"," pandas ",[174,209,210],{"class":202},"as",[174,212,213],{"class":206}," pd\n",[174,215,217],{"class":75,"line":216},2,[174,218,220],{"emptyLinePlaceholder":219},true,"\n",[174,222,224,227,230],{"class":75,"line":223},3,[174,225,226],{"class":206},"orders ",[174,228,229],{"class":202},"=",[174,231,232],{"class":206}," pd.DataFrame({\n",[174,234,236,239,242,246,249,252,254,257,259,262,264,267],{"class":75,"line":235},4,[174,237,238],{"class":183},"    \"Order\"",[174,240,241],{"class":206},": [",[174,243,245],{"class":244},"sP0c6","1001",[174,247,248],{"class":206},", ",[174,250,251],{"class":244},"1002",[174,253,248],{"class":206},[174,255,256],{"class":244},"1003",[174,258,248],{"class":206},[174,260,261],{"class":244},"1004",[174,263,248],{"class":206},[174,265,266],{"class":244},"1005",[174,268,269],{"class":206},"],\n",[174,271,273,276,278,281,283,286,288,290,292,295,297,300],{"class":75,"line":272},5,[174,274,275],{"class":183},"    \"SKU\"",[174,277,241],{"class":206},[174,279,280],{"class":183},"\"A-100\"",[174,282,248],{"class":206},[174,284,285],{"class":183},"\"B-200\"",[174,287,248],{"class":206},[174,289,280],{"class":183},[174,291,248],{"class":206},[174,293,294],{"class":183},"\"C-300\"",[174,296,248],{"class":206},[174,298,299],{"class":183},"\"D-400\"",[174,301,269],{"class":206},[174,303,305,308,310,313,315,318,320,323,325,327,329,331],{"class":75,"line":304},6,[174,306,307],{"class":183},"    \"Region\"",[174,309,241],{"class":206},[174,311,312],{"class":183},"\"North\"",[174,314,248],{"class":206},[174,316,317],{"class":183},"\"South\"",[174,319,248],{"class":206},[174,321,322],{"class":183},"\"West\"",[174,324,248],{"class":206},[174,326,312],{"class":183},[174,328,248],{"class":206},[174,330,317],{"class":183},[174,332,269],{"class":206},[174,334,336,339,341,344,346,349,351,354,356,359,361,364],{"class":75,"line":335},7,[174,337,338],{"class":183},"    \"Units\"",[174,340,241],{"class":206},[174,342,343],{"class":244},"12",[174,345,248],{"class":206},[174,347,348],{"class":244},"4",[174,350,248],{"class":206},[174,352,353],{"class":244},"9",[174,355,248],{"class":206},[174,357,358],{"class":244},"22",[174,360,248],{"class":206},[174,362,363],{"class":244},"7",[174,365,269],{"class":206},[174,367,369],{"class":75,"line":368},8,[174,370,371],{"class":206},"})\n",[174,373,375],{"class":75,"line":374},9,[174,376,220],{"emptyLinePlaceholder":219},[174,378,380,383,385],{"class":75,"line":379},10,[174,381,382],{"class":206},"products ",[174,384,229],{"class":202},[174,386,232],{"class":206},[174,388,390,392,394,396,398,400,402,404],{"class":75,"line":389},11,[174,391,275],{"class":183},[174,393,241],{"class":206},[174,395,280],{"class":183},[174,397,248],{"class":206},[174,399,285],{"class":183},[174,401,248],{"class":206},[174,403,294],{"class":183},[174,405,269],{"class":206},[174,407,409,412,414,417,419,422,424,427],{"class":75,"line":408},12,[174,410,411],{"class":183},"    \"Description\"",[174,413,241],{"class":206},[174,415,416],{"class":183},"\"Widget, small\"",[174,418,248],{"class":206},[174,420,421],{"class":183},"\"Gadget\"",[174,423,248],{"class":206},[174,425,426],{"class":183},"\"Widget, large\"",[174,428,269],{"class":206},[174,430,432,435,437,440,442,445,447,450],{"class":75,"line":431},13,[174,433,434],{"class":183},"    \"Unit_Price\"",[174,436,241],{"class":206},[174,438,439],{"class":244},"19.99",[174,441,248],{"class":206},[174,443,444],{"class":244},"49.50",[174,446,248],{"class":206},[174,448,449],{"class":244},"34.75",[174,451,269],{"class":206},[174,453,455,458,460,463,465,468,470,472],{"class":75,"line":454},14,[174,456,457],{"class":183},"    \"Category\"",[174,459,241],{"class":206},[174,461,462],{"class":183},"\"Widgets\"",[174,464,248],{"class":206},[174,466,467],{"class":183},"\"Gadgets\"",[174,469,248],{"class":206},[174,471,462],{"class":183},[174,473,269],{"class":206},[174,475,477],{"class":75,"line":476},15,[174,478,371],{"class":206},[160,480,482],{"id":481},"the-one-column-form-map","The one-column form: map",[10,484,485,486,488],{},"When the lookup returns a single column and the key is a single column, ",[14,487,20],{}," is the shortest\npossible translation and it cannot change the shape of your data.",[165,490,492],{"className":193,"code":491,"language":195,"meta":170,"style":170},"# =INDEX(products!C:C, MATCH(B2, products!A:A, 0))\nprices = products.set_index(\"SKU\")[\"Unit_Price\"]\norders[\"Unit_Price\"] = orders[\"SKU\"].map(prices)\n\norders[\"Line_Total\"] = orders[\"Units\"] * orders[\"Unit_Price\"]\nprint(orders)\n",[14,493,494,500,522,542,546,573],{"__ignoreMap":170},[174,495,496],{"class":75,"line":176},[174,497,499],{"class":498},"s-wDw","# =INDEX(products!C:C, MATCH(B2, products!A:A, 0))\n",[174,501,502,505,507,510,513,516,519],{"class":75,"line":216},[174,503,504],{"class":206},"prices ",[174,506,229],{"class":202},[174,508,509],{"class":206}," products.set_index(",[174,511,512],{"class":183},"\"SKU\"",[174,514,515],{"class":206},")[",[174,517,518],{"class":183},"\"Unit_Price\"",[174,520,521],{"class":206},"]\n",[174,523,524,527,529,532,534,537,539],{"class":75,"line":223},[174,525,526],{"class":206},"orders[",[174,528,518],{"class":183},[174,530,531],{"class":206},"] ",[174,533,229],{"class":202},[174,535,536],{"class":206}," orders[",[174,538,512],{"class":183},[174,540,541],{"class":206},"].map(prices)\n",[174,543,544],{"class":75,"line":235},[174,545,220],{"emptyLinePlaceholder":219},[174,547,548,550,553,555,557,559,562,564,567,569,571],{"class":75,"line":272},[174,549,526],{"class":206},[174,551,552],{"class":183},"\"Line_Total\"",[174,554,531],{"class":206},[174,556,229],{"class":202},[174,558,536],{"class":206},[174,560,561],{"class":183},"\"Units\"",[174,563,531],{"class":206},[174,565,566],{"class":202},"*",[174,568,536],{"class":206},[174,570,518],{"class":183},[174,572,521],{"class":206},[174,574,575,578],{"class":75,"line":304},[174,576,577],{"class":244},"print",[174,579,580],{"class":206},"(orders)\n",[10,582,583,586,587,589,590,593],{},[14,584,585],{},"set_index(\"SKU\")[\"Unit_Price\"]"," produces a Series indexed by SKU — which is exactly what a lookup\ntable is. ",[14,588,20],{}," then translates each SKU into its price, leaving NaN where the SKU is unknown. The\n",[14,591,592],{},"D-400"," row demonstrates that: no product record, so no price, and the line total is NaN rather than\na silently wrong number.",[10,595,596,598],{},[14,597,20],{}," also accepts a plain dictionary, which is often the clearest form when the mapping is short and\nbelongs in the code rather than in a file:",[165,600,602],{"className":193,"code":601,"language":195,"meta":170,"style":170},"region_owner = {\"North\": \"Ana\", \"South\": \"Ben\", \"West\": \"Dev\"}\norders[\"Owner\"] = orders[\"Region\"].map(region_owner)\n",[14,603,604,643],{"__ignoreMap":170},[174,605,606,609,611,614,616,619,622,624,626,628,631,633,635,637,640],{"class":75,"line":176},[174,607,608],{"class":206},"region_owner ",[174,610,229],{"class":202},[174,612,613],{"class":206}," {",[174,615,312],{"class":183},[174,617,618],{"class":206},": ",[174,620,621],{"class":183},"\"Ana\"",[174,623,248],{"class":206},[174,625,317],{"class":183},[174,627,618],{"class":206},[174,629,630],{"class":183},"\"Ben\"",[174,632,248],{"class":206},[174,634,322],{"class":183},[174,636,618],{"class":206},[174,638,639],{"class":183},"\"Dev\"",[174,641,642],{"class":206},"}\n",[174,644,645,647,650,652,654,656,659],{"class":75,"line":216},[174,646,526],{"class":206},[174,648,649],{"class":183},"\"Owner\"",[174,651,531],{"class":206},[174,653,229],{"class":202},[174,655,536],{"class":206},[174,657,658],{"class":183},"\"Region\"",[174,660,661],{"class":206},"].map(region_owner)\n",[160,663,665],{"id":664},"the-many-column-form-merge","The many-column form: merge",[10,667,668,669,671],{},"When the lookup should bring back several columns — which in Excel means one INDEX\u002FMATCH per column,\neach re-scanning the table — ",[14,670,16],{}," does it in one call.",[165,673,675],{"className":193,"code":674,"language":195,"meta":170,"style":170},"# Three INDEX\u002FMATCH formulas, replaced by one join\nenriched = orders.merge(\n    products[[\"SKU\", \"Description\", \"Unit_Price\", \"Category\"]],\n    on=\"SKU\",\n    how=\"left\",\n    validate=\"m:1\",\n)\nprint(enriched)\n",[14,676,677,682,692,716,729,741,753,758],{"__ignoreMap":170},[174,678,679],{"class":75,"line":176},[174,680,681],{"class":498},"# Three INDEX\u002FMATCH formulas, replaced by one join\n",[174,683,684,687,689],{"class":75,"line":216},[174,685,686],{"class":206},"enriched ",[174,688,229],{"class":202},[174,690,691],{"class":206}," orders.merge(\n",[174,693,694,697,699,701,704,706,708,710,713],{"class":75,"line":223},[174,695,696],{"class":206},"    products[[",[174,698,512],{"class":183},[174,700,248],{"class":206},[174,702,703],{"class":183},"\"Description\"",[174,705,248],{"class":206},[174,707,518],{"class":183},[174,709,248],{"class":206},[174,711,712],{"class":183},"\"Category\"",[174,714,715],{"class":206},"]],\n",[174,717,718,722,724,726],{"class":75,"line":235},[174,719,721],{"class":720},"sa561","    on",[174,723,229],{"class":202},[174,725,512],{"class":183},[174,727,728],{"class":206},",\n",[174,730,731,734,736,739],{"class":75,"line":272},[174,732,733],{"class":720},"    how",[174,735,229],{"class":202},[174,737,738],{"class":183},"\"left\"",[174,740,728],{"class":206},[174,742,743,746,748,751],{"class":75,"line":304},[174,744,745],{"class":720},"    validate",[174,747,229],{"class":202},[174,749,750],{"class":183},"\"m:1\"",[174,752,728],{"class":206},[174,754,755],{"class":75,"line":335},[174,756,757],{"class":206},")\n",[174,759,760,762],{"class":75,"line":368},[174,761,577],{"class":244},[174,763,764],{"class":206},"(enriched)\n",[10,766,767,770,771,774,775,779],{},[14,768,769],{},"how=\"left\""," gives VLOOKUP's semantics: every order survives, unmatched ones get NaN.\n",[14,772,773],{},"validate=\"m:1\""," is the argument worth adopting as a habit — it asserts that the lookup table has at\nmost one row per key, and raises immediately if it does not. Without it, a duplicated key in the\nlookup table silently multiplies rows, and the total at the bottom of the report is quietly too\nlarge. That failure is described in full in\n",[23,776,778],{"href":777},"\u002Fadvanced-data-transformation-and-cleaning\u002Fmerging-and-joining-excel-dataframes\u002Fvlookup-equivalent-in-pandas-for-excel-files\u002F","VLOOKUP Equivalent in pandas for Excel Files",".",[160,781,783],{"id":782},"reporting-what-did-not-match","Reporting what did not match",[10,785,786],{},"Excel wraps the formula in IFERROR and shows a blank. That hides the problem one cell at a time;\npandas lets you count it once.",[165,788,790],{"className":193,"code":789,"language":195,"meta":170,"style":170},"unmatched = enriched.loc[enriched[\"Description\"].isna(), [\"Order\", \"SKU\"]]\nif not unmatched.empty:\n    print(f\"{len(unmatched)} order(s) reference an unknown SKU:\")\n    print(unmatched.to_string(index=False))\n",[14,791,792,817,828,860],{"__ignoreMap":170},[174,793,794,797,799,802,804,807,810,812,814],{"class":75,"line":176},[174,795,796],{"class":206},"unmatched ",[174,798,229],{"class":202},[174,800,801],{"class":206}," enriched.loc[enriched[",[174,803,703],{"class":183},[174,805,806],{"class":206},"].isna(), [",[174,808,809],{"class":183},"\"Order\"",[174,811,248],{"class":206},[174,813,512],{"class":183},[174,815,816],{"class":206},"]]\n",[174,818,819,822,825],{"class":75,"line":216},[174,820,821],{"class":202},"if",[174,823,824],{"class":202}," not",[174,826,827],{"class":206}," unmatched.empty:\n",[174,829,830,833,836,839,842,846,849,852,855,858],{"class":75,"line":223},[174,831,832],{"class":244},"    print",[174,834,835],{"class":206},"(",[174,837,838],{"class":202},"f",[174,840,841],{"class":183},"\"",[174,843,845],{"class":844},"sSjpA","{",[174,847,848],{"class":244},"len",[174,850,851],{"class":206},"(unmatched)",[174,853,854],{"class":844},"}",[174,856,857],{"class":183}," order(s) reference an unknown SKU:\"",[174,859,757],{"class":206},[174,861,862,864,867,870,872,875],{"class":75,"line":235},[174,863,832],{"class":244},[174,865,866],{"class":206},"(unmatched.to_string(",[174,868,869],{"class":720},"index",[174,871,229],{"class":202},[174,873,874],{"class":244},"False",[174,876,877],{"class":206},"))\n",[10,879,880],{},"An unmatched key is nearly always a data problem — a product retired without updating the catalogue,\na typo in an export, a trailing space — and it is worth reporting rather than filling. When a default\ngenuinely is correct, make it explicit rather than incidental:",[165,882,884],{"className":193,"code":883,"language":195,"meta":170,"style":170},"enriched[\"Category\"] = enriched[\"Category\"].fillna(\"Uncategorised\")\n",[14,885,886],{"__ignoreMap":170},[174,887,888,891,893,895,897,900,902,905,908],{"class":75,"line":176},[174,889,890],{"class":206},"enriched[",[174,892,712],{"class":183},[174,894,531],{"class":206},[174,896,229],{"class":202},[174,898,899],{"class":206}," enriched[",[174,901,712],{"class":183},[174,903,904],{"class":206},"].fillna(",[174,906,907],{"class":183},"\"Uncategorised\"",[174,909,757],{"class":206},[160,911,913],{"id":912},"composite-keys","Composite keys",[33,915,42,920,42,923,42,926,42,929,42,933,42,939,42,945,42,950,42,955,42,958,42,961,42,964,42,967,42,971,42,974,42,977,42,982,42,985],{"viewBox":916,"role":36,"ariaLabelledBy":917,"xmlns":40,"style":41},"0 0 760 232",[918,919],"im-composite-t","im-composite-d",[44,921,922],{"id":918},"Matching on two columns without a helper column",[48,924,925],{"id":919},"Excel needs a concatenated helper column or an array formula to match on two keys, while merge takes a list of column names and compares them as a tuple.",[52,927],{"x":54,"y":54,"width":55,"height":928,"fill":57},"232",[69,930,116],{"x":133,"y":931,"style":932},"32","font-size:13px;font-weight:600;fill:var(--muted,#5b6780);text-anchor:middle",[52,934],{"x":935,"y":936,"width":937,"height":938,"rx":343,"fill":101,"stroke":102,"style":67},"24.0","74","208.0","96",[69,940,944],{"x":941,"y":942,"style":943},"128.0","114","font-size:14px;font-weight:700;fill:var(--brand-strong,#4338ca);text-anchor:middle","two key columns",[69,946,949],{"x":941,"y":947,"style":948},"136","font-size:11.5px;font-weight:400;fill:var(--muted,#5b6780);text-anchor:middle","Region and Category",[75,951],{"x1":952,"y1":953,"x2":954,"y2":953,"stroke":102,"style":67},"237.0","122.0","269.0",[143,956],{"points":957,"fill":146},"269.0,122.0 260.0,117.0 260.0,127.0",[52,959],{"x":960,"y":936,"width":937,"height":938,"rx":343,"fill":101,"stroke":102,"style":67},"276.0",[69,962,963],{"x":133,"y":942,"style":943},"on=[...]",[69,965,966],{"x":133,"y":947,"style":948},"compared as a pair",[75,968],{"x1":969,"y1":953,"x2":970,"y2":953,"stroke":102,"style":67},"489.0","521.0",[143,972],{"points":973,"fill":146},"521.0,122.0 512.0,117.0 512.0,127.0",[52,975],{"x":976,"y":936,"width":937,"height":938,"rx":343,"fill":65,"stroke":66,"style":67},"528.0",[69,978,981],{"x":979,"y":942,"style":980},"632.0","font-size:14px;font-weight:700;fill:var(--teal-ink,#0b6157);text-anchor:middle","joined rows",[69,983,984],{"x":979,"y":947,"style":948},"no helper column",[69,986,987],{"x":133,"y":156,"style":157},"concatenating keys by hand can collide; a column list cannot",[10,989,990,991,993],{},"Matching on two columns is where the spreadsheet version gets ugly: Excel needs a helper column\nconcatenating the keys, or an array formula. ",[14,992,16],{}," takes a list.",[165,995,997],{"className":193,"code":996,"language":195,"meta":170,"style":170},"targets = pd.DataFrame({\n    \"Region\": [\"North\", \"North\", \"South\", \"South\"],\n    \"Category\": [\"Widgets\", \"Gadgets\", \"Widgets\", \"Gadgets\"],\n    \"Target\": [40000.0, 15000.0, 22000.0, 30000.0],\n})\n\nwith_targets = enriched.merge(targets, on=[\"Region\", \"Category\"], how=\"left\")\nprint(with_targets[[\"Order\", \"Region\", \"Category\", \"Target\"]])\n",[14,998,999,1008,1030,1052,1079,1083,1087,1123],{"__ignoreMap":170},[174,1000,1001,1004,1006],{"class":75,"line":176},[174,1002,1003],{"class":206},"targets ",[174,1005,229],{"class":202},[174,1007,232],{"class":206},[174,1009,1010,1012,1014,1016,1018,1020,1022,1024,1026,1028],{"class":75,"line":216},[174,1011,307],{"class":183},[174,1013,241],{"class":206},[174,1015,312],{"class":183},[174,1017,248],{"class":206},[174,1019,312],{"class":183},[174,1021,248],{"class":206},[174,1023,317],{"class":183},[174,1025,248],{"class":206},[174,1027,317],{"class":183},[174,1029,269],{"class":206},[174,1031,1032,1034,1036,1038,1040,1042,1044,1046,1048,1050],{"class":75,"line":223},[174,1033,457],{"class":183},[174,1035,241],{"class":206},[174,1037,462],{"class":183},[174,1039,248],{"class":206},[174,1041,467],{"class":183},[174,1043,248],{"class":206},[174,1045,462],{"class":183},[174,1047,248],{"class":206},[174,1049,467],{"class":183},[174,1051,269],{"class":206},[174,1053,1054,1057,1059,1062,1064,1067,1069,1072,1074,1077],{"class":75,"line":235},[174,1055,1056],{"class":183},"    \"Target\"",[174,1058,241],{"class":206},[174,1060,1061],{"class":244},"40000.0",[174,1063,248],{"class":206},[174,1065,1066],{"class":244},"15000.0",[174,1068,248],{"class":206},[174,1070,1071],{"class":244},"22000.0",[174,1073,248],{"class":206},[174,1075,1076],{"class":244},"30000.0",[174,1078,269],{"class":206},[174,1080,1081],{"class":75,"line":272},[174,1082,371],{"class":206},[174,1084,1085],{"class":75,"line":304},[174,1086,220],{"emptyLinePlaceholder":219},[174,1088,1089,1092,1094,1097,1100,1102,1105,1107,1109,1111,1114,1117,1119,1121],{"class":75,"line":335},[174,1090,1091],{"class":206},"with_targets ",[174,1093,229],{"class":202},[174,1095,1096],{"class":206}," enriched.merge(targets, ",[174,1098,1099],{"class":720},"on",[174,1101,229],{"class":202},[174,1103,1104],{"class":206},"[",[174,1106,658],{"class":183},[174,1108,248],{"class":206},[174,1110,712],{"class":183},[174,1112,1113],{"class":206},"], ",[174,1115,1116],{"class":720},"how",[174,1118,229],{"class":202},[174,1120,738],{"class":183},[174,1122,757],{"class":206},[174,1124,1125,1127,1130,1132,1134,1136,1138,1140,1142,1145],{"class":75,"line":368},[174,1126,577],{"class":244},[174,1128,1129],{"class":206},"(with_targets[[",[174,1131,809],{"class":183},[174,1133,248],{"class":206},[174,1135,658],{"class":183},[174,1137,248],{"class":206},[174,1139,712],{"class":183},[174,1141,248],{"class":206},[174,1143,1144],{"class":183},"\"Target\"",[174,1146,1147],{"class":206},"]])\n",[10,1149,1150,1151,1154,1155,1158],{},"No helper column, no concatenation, and no risk that ",[14,1152,1153],{},"\"North\" + \"Widgets\""," collides with\n",[14,1156,1157],{},"\"NorthWid\" + \"gets\""," — which is a real failure mode of the concatenation trick when keys have\nvariable length.",[10,1160,1161,1162,1165,1166,1169],{},"When the two frames name the same concept differently, ",[14,1163,1164],{},"left_on"," and ",[14,1167,1168],{},"right_on"," avoid renaming\nanything:",[165,1171,1173],{"className":193,"code":1172,"language":195,"meta":170,"style":170},"merged = orders.merge(\n    products.rename(columns={\"SKU\": \"Item_Code\"}),\n    left_on=\"SKU\", right_on=\"Item_Code\", how=\"left\",\n).drop(columns=\"Item_Code\")\n",[14,1174,1175,1184,1206,1233],{"__ignoreMap":170},[174,1176,1177,1180,1182],{"class":75,"line":176},[174,1178,1179],{"class":206},"merged ",[174,1181,229],{"class":202},[174,1183,691],{"class":206},[174,1185,1186,1189,1192,1194,1196,1198,1200,1203],{"class":75,"line":216},[174,1187,1188],{"class":206},"    products.rename(",[174,1190,1191],{"class":720},"columns",[174,1193,229],{"class":202},[174,1195,845],{"class":206},[174,1197,512],{"class":183},[174,1199,618],{"class":206},[174,1201,1202],{"class":183},"\"Item_Code\"",[174,1204,1205],{"class":206},"}),\n",[174,1207,1208,1211,1213,1215,1217,1219,1221,1223,1225,1227,1229,1231],{"class":75,"line":223},[174,1209,1210],{"class":720},"    left_on",[174,1212,229],{"class":202},[174,1214,512],{"class":183},[174,1216,248],{"class":206},[174,1218,1168],{"class":720},[174,1220,229],{"class":202},[174,1222,1202],{"class":183},[174,1224,248],{"class":206},[174,1226,1116],{"class":720},[174,1228,229],{"class":202},[174,1230,738],{"class":183},[174,1232,728],{"class":206},[174,1234,1235,1238,1240,1242,1244],{"class":75,"line":235},[174,1236,1237],{"class":206},").drop(",[174,1239,1191],{"class":720},[174,1241,229],{"class":202},[174,1243,1202],{"class":183},[174,1245,757],{"class":206},[160,1247,1249],{"id":1248},"approximate-matches-merge_asof","Approximate matches: merge_asof",[10,1251,1252,1253,1255],{},"MATCH with a match type of 1 finds the largest value less than or equal to the lookup — the mechanism\nbehind every rate table, tax band and volume-discount tier. ",[14,1254,30],{}," is the direct equivalent and\nit handles the sorting requirement explicitly.",[165,1257,1259],{"className":193,"code":1258,"language":195,"meta":170,"style":170},"discounts = pd.DataFrame({\n    \"Min_Units\": [0, 10, 20],\n    \"Discount\": [0.0, 0.05, 0.12],\n})\n\npriced = pd.merge_asof(\n    orders.sort_values(\"Units\"),\n    discounts.sort_values(\"Min_Units\"),\n    left_on=\"Units\",\n    right_on=\"Min_Units\",\n    direction=\"backward\",\n)\nprint(priced[[\"Order\", \"Units\", \"Min_Units\", \"Discount\"]])\n",[14,1260,1261,1270,1290,1312,1316,1320,1330,1340,1350,1360,1371,1383,1387],{"__ignoreMap":170},[174,1262,1263,1266,1268],{"class":75,"line":176},[174,1264,1265],{"class":206},"discounts ",[174,1267,229],{"class":202},[174,1269,232],{"class":206},[174,1271,1272,1275,1277,1279,1281,1284,1286,1288],{"class":75,"line":216},[174,1273,1274],{"class":183},"    \"Min_Units\"",[174,1276,241],{"class":206},[174,1278,54],{"class":244},[174,1280,248],{"class":206},[174,1282,1283],{"class":244},"10",[174,1285,248],{"class":206},[174,1287,60],{"class":244},[174,1289,269],{"class":206},[174,1291,1292,1295,1297,1300,1302,1305,1307,1310],{"class":75,"line":223},[174,1293,1294],{"class":183},"    \"Discount\"",[174,1296,241],{"class":206},[174,1298,1299],{"class":244},"0.0",[174,1301,248],{"class":206},[174,1303,1304],{"class":244},"0.05",[174,1306,248],{"class":206},[174,1308,1309],{"class":244},"0.12",[174,1311,269],{"class":206},[174,1313,1314],{"class":75,"line":235},[174,1315,371],{"class":206},[174,1317,1318],{"class":75,"line":272},[174,1319,220],{"emptyLinePlaceholder":219},[174,1321,1322,1325,1327],{"class":75,"line":304},[174,1323,1324],{"class":206},"priced ",[174,1326,229],{"class":202},[174,1328,1329],{"class":206}," pd.merge_asof(\n",[174,1331,1332,1335,1337],{"class":75,"line":335},[174,1333,1334],{"class":206},"    orders.sort_values(",[174,1336,561],{"class":183},[174,1338,1339],{"class":206},"),\n",[174,1341,1342,1345,1348],{"class":75,"line":368},[174,1343,1344],{"class":206},"    discounts.sort_values(",[174,1346,1347],{"class":183},"\"Min_Units\"",[174,1349,1339],{"class":206},[174,1351,1352,1354,1356,1358],{"class":75,"line":374},[174,1353,1210],{"class":720},[174,1355,229],{"class":202},[174,1357,561],{"class":183},[174,1359,728],{"class":206},[174,1361,1362,1365,1367,1369],{"class":75,"line":379},[174,1363,1364],{"class":720},"    right_on",[174,1366,229],{"class":202},[174,1368,1347],{"class":183},[174,1370,728],{"class":206},[174,1372,1373,1376,1378,1381],{"class":75,"line":389},[174,1374,1375],{"class":720},"    direction",[174,1377,229],{"class":202},[174,1379,1380],{"class":183},"\"backward\"",[174,1382,728],{"class":206},[174,1384,1385],{"class":75,"line":408},[174,1386,757],{"class":206},[174,1388,1389,1391,1394,1396,1398,1400,1402,1404,1406,1409],{"class":75,"line":431},[174,1390,577],{"class":244},[174,1392,1393],{"class":206},"(priced[[",[174,1395,809],{"class":183},[174,1397,248],{"class":206},[174,1399,561],{"class":183},[174,1401,248],{"class":206},[174,1403,1347],{"class":183},[174,1405,248],{"class":206},[174,1407,1408],{"class":183},"\"Discount\"",[174,1410,1147],{"class":206},[10,1412,1413,1414,1417,1418,1165,1421,1424],{},"Both frames must be sorted on the join key or the result is wrong rather than an error — the same\nrequirement Excel's approximate MATCH imposes, and the same silent failure when it is not met.\n",[14,1415,1416],{},"direction=\"backward\""," is the default and matches Excel's behaviour; ",[14,1419,1420],{},"\"forward\"",[14,1422,1423],{},"\"nearest\""," have\nno spreadsheet equivalent at all.",[160,1426,1428],{"id":1427},"two-dimensional-lookups","Two-dimensional lookups",[10,1430,1431],{},"The other classic INDEX\u002FMATCH shape uses two MATCH calls — one for the row, one for the column — to\npick a value out of a rectangular grid: a rate by region and month, a price by size and finish. That\nlayout is a pivot table stored as a sheet, and the pandas answer is to unpivot it back into rows\nbefore joining.",[165,1433,1435],{"className":193,"code":1434,"language":195,"meta":170,"style":170},"grid = pd.DataFrame({\n    \"Region\": [\"North\", \"South\", \"West\"],\n    \"Jan\": [0.10, 0.08, 0.06],\n    \"Feb\": [0.11, 0.08, 0.07],\n    \"Mar\": [0.12, 0.09, 0.07],\n})\n\nrates = grid.melt(id_vars=\"Region\", var_name=\"Month\", value_name=\"Rate\")\nprint(rates.head())\n\norders[\"Month\"] = [\"Jan\", \"Feb\", \"Feb\", \"Mar\", \"Mar\"]\nwith_rate = orders.merge(rates, on=[\"Region\", \"Month\"], how=\"left\")\n",[14,1436,1437,1446,1464,1486,1507,1527,1531,1535,1574,1581,1585,1621],{"__ignoreMap":170},[174,1438,1439,1442,1444],{"class":75,"line":176},[174,1440,1441],{"class":206},"grid ",[174,1443,229],{"class":202},[174,1445,232],{"class":206},[174,1447,1448,1450,1452,1454,1456,1458,1460,1462],{"class":75,"line":216},[174,1449,307],{"class":183},[174,1451,241],{"class":206},[174,1453,312],{"class":183},[174,1455,248],{"class":206},[174,1457,317],{"class":183},[174,1459,248],{"class":206},[174,1461,322],{"class":183},[174,1463,269],{"class":206},[174,1465,1466,1469,1471,1474,1476,1479,1481,1484],{"class":75,"line":223},[174,1467,1468],{"class":183},"    \"Jan\"",[174,1470,241],{"class":206},[174,1472,1473],{"class":244},"0.10",[174,1475,248],{"class":206},[174,1477,1478],{"class":244},"0.08",[174,1480,248],{"class":206},[174,1482,1483],{"class":244},"0.06",[174,1485,269],{"class":206},[174,1487,1488,1491,1493,1496,1498,1500,1502,1505],{"class":75,"line":235},[174,1489,1490],{"class":183},"    \"Feb\"",[174,1492,241],{"class":206},[174,1494,1495],{"class":244},"0.11",[174,1497,248],{"class":206},[174,1499,1478],{"class":244},[174,1501,248],{"class":206},[174,1503,1504],{"class":244},"0.07",[174,1506,269],{"class":206},[174,1508,1509,1512,1514,1516,1518,1521,1523,1525],{"class":75,"line":272},[174,1510,1511],{"class":183},"    \"Mar\"",[174,1513,241],{"class":206},[174,1515,1309],{"class":244},[174,1517,248],{"class":206},[174,1519,1520],{"class":244},"0.09",[174,1522,248],{"class":206},[174,1524,1504],{"class":244},[174,1526,269],{"class":206},[174,1528,1529],{"class":75,"line":304},[174,1530,371],{"class":206},[174,1532,1533],{"class":75,"line":335},[174,1534,220],{"emptyLinePlaceholder":219},[174,1536,1537,1540,1542,1545,1548,1550,1552,1554,1557,1559,1562,1564,1567,1569,1572],{"class":75,"line":368},[174,1538,1539],{"class":206},"rates ",[174,1541,229],{"class":202},[174,1543,1544],{"class":206}," grid.melt(",[174,1546,1547],{"class":720},"id_vars",[174,1549,229],{"class":202},[174,1551,658],{"class":183},[174,1553,248],{"class":206},[174,1555,1556],{"class":720},"var_name",[174,1558,229],{"class":202},[174,1560,1561],{"class":183},"\"Month\"",[174,1563,248],{"class":206},[174,1565,1566],{"class":720},"value_name",[174,1568,229],{"class":202},[174,1570,1571],{"class":183},"\"Rate\"",[174,1573,757],{"class":206},[174,1575,1576,1578],{"class":75,"line":374},[174,1577,577],{"class":244},[174,1579,1580],{"class":206},"(rates.head())\n",[174,1582,1583],{"class":75,"line":379},[174,1584,220],{"emptyLinePlaceholder":219},[174,1586,1587,1589,1591,1593,1595,1598,1601,1603,1606,1608,1610,1612,1615,1617,1619],{"class":75,"line":389},[174,1588,526],{"class":206},[174,1590,1561],{"class":183},[174,1592,531],{"class":206},[174,1594,229],{"class":202},[174,1596,1597],{"class":206}," [",[174,1599,1600],{"class":183},"\"Jan\"",[174,1602,248],{"class":206},[174,1604,1605],{"class":183},"\"Feb\"",[174,1607,248],{"class":206},[174,1609,1605],{"class":183},[174,1611,248],{"class":206},[174,1613,1614],{"class":183},"\"Mar\"",[174,1616,248],{"class":206},[174,1618,1614],{"class":183},[174,1620,521],{"class":206},[174,1622,1623,1626,1628,1631,1633,1635,1637,1639,1641,1643,1645,1647,1649,1651],{"class":75,"line":408},[174,1624,1625],{"class":206},"with_rate ",[174,1627,229],{"class":202},[174,1629,1630],{"class":206}," orders.merge(rates, ",[174,1632,1099],{"class":720},[174,1634,229],{"class":202},[174,1636,1104],{"class":206},[174,1638,658],{"class":183},[174,1640,248],{"class":206},[174,1642,1561],{"class":183},[174,1644,1113],{"class":206},[174,1646,1116],{"class":720},[174,1648,229],{"class":202},[174,1650,738],{"class":183},[174,1652,757],{"class":206},[10,1654,1655,1658,1659,1663],{},[14,1656,1657],{},"melt"," turns the wide grid into one row per Region-Month pair, after which the two-dimensional lookup\nis an ordinary composite-key join. That reshaping step is worth doing even when a direct lookup would\nwork, because the long form is the shape every other pandas operation expects — and it survives a new\nmonth being added to the grid, which a formula referencing a fixed column range does not.\n",[23,1660,1662],{"href":1661},"\u002Fadvanced-data-transformation-and-cleaning\u002Fcreating-pivot-tables-from-excel-data\u002Funpivot-a-wide-excel-sheet-with-pandas-melt\u002F","Unpivot a Wide Excel Sheet with pandas melt","\ncovers the reshaping in detail.",[160,1665,1667],{"id":1666},"keeping-the-lookup-table-honest","Keeping the lookup table honest",[10,1669,1670,1671,1673],{},"A lookup is only as reliable as the table behind it, and two checks catch most of what goes wrong.\nThe first is uniqueness of the key, which ",[14,1672,773],{}," enforces at merge time. The second is\ncoverage — whether every key in the data has a row in the table — which nothing enforces unless you\nask.",[165,1675,1677],{"className":193,"code":1676,"language":195,"meta":170,"style":170},"def check_lookup(frame, table, key):\n    duplicated = table[key].duplicated().sum()\n    uncovered = sorted(set(frame[key].dropna()) - set(table[key]))\n    if duplicated:\n        raise ValueError(f\"lookup table has {duplicated} duplicate {key} value(s)\")\n    if uncovered:\n        print(f\"warning: {len(uncovered)} unmatched {key}(s): {uncovered[:10]}\")\n\ncheck_lookup(orders, products, \"SKU\")\n",[14,1678,1679,1691,1701,1728,1736,1773,1780,1829,1833],{"__ignoreMap":170},[174,1680,1681,1684,1688],{"class":75,"line":176},[174,1682,1683],{"class":202},"def",[174,1685,1687],{"class":1686},"s_Opv"," check_lookup",[174,1689,1690],{"class":206},"(frame, table, key):\n",[174,1692,1693,1696,1698],{"class":75,"line":216},[174,1694,1695],{"class":206},"    duplicated ",[174,1697,229],{"class":202},[174,1699,1700],{"class":206}," table[key].duplicated().sum()\n",[174,1702,1703,1706,1708,1711,1713,1716,1719,1722,1725],{"class":75,"line":223},[174,1704,1705],{"class":206},"    uncovered ",[174,1707,229],{"class":202},[174,1709,1710],{"class":244}," sorted",[174,1712,835],{"class":206},[174,1714,1715],{"class":244},"set",[174,1717,1718],{"class":206},"(frame[key].dropna()) ",[174,1720,1721],{"class":202},"-",[174,1723,1724],{"class":244}," set",[174,1726,1727],{"class":206},"(table[key]))\n",[174,1729,1730,1733],{"class":75,"line":235},[174,1731,1732],{"class":202},"    if",[174,1734,1735],{"class":206}," duplicated:\n",[174,1737,1738,1741,1744,1746,1748,1751,1753,1756,1758,1761,1763,1766,1768,1771],{"class":75,"line":272},[174,1739,1740],{"class":202},"        raise",[174,1742,1743],{"class":244}," ValueError",[174,1745,835],{"class":206},[174,1747,838],{"class":202},[174,1749,1750],{"class":183},"\"lookup table has ",[174,1752,845],{"class":844},[174,1754,1755],{"class":206},"duplicated",[174,1757,854],{"class":844},[174,1759,1760],{"class":183}," duplicate ",[174,1762,845],{"class":844},[174,1764,1765],{"class":206},"key",[174,1767,854],{"class":844},[174,1769,1770],{"class":183}," value(s)\"",[174,1772,757],{"class":206},[174,1774,1775,1777],{"class":75,"line":304},[174,1776,1732],{"class":202},[174,1778,1779],{"class":206}," uncovered:\n",[174,1781,1782,1785,1787,1789,1792,1794,1796,1799,1801,1804,1806,1808,1810,1813,1815,1818,1820,1823,1825,1827],{"class":75,"line":335},[174,1783,1784],{"class":244},"        print",[174,1786,835],{"class":206},[174,1788,838],{"class":202},[174,1790,1791],{"class":183},"\"warning: ",[174,1793,845],{"class":844},[174,1795,848],{"class":244},[174,1797,1798],{"class":206},"(uncovered)",[174,1800,854],{"class":844},[174,1802,1803],{"class":183}," unmatched ",[174,1805,845],{"class":844},[174,1807,1765],{"class":206},[174,1809,854],{"class":844},[174,1811,1812],{"class":183},"(s): ",[174,1814,845],{"class":844},[174,1816,1817],{"class":206},"uncovered[:",[174,1819,1283],{"class":244},[174,1821,1822],{"class":206},"]",[174,1824,854],{"class":844},[174,1826,841],{"class":183},[174,1828,757],{"class":206},[174,1830,1831],{"class":75,"line":368},[174,1832,220],{"emptyLinePlaceholder":219},[174,1834,1835,1838,1840],{"class":75,"line":374},[174,1836,1837],{"class":206},"check_lookup(orders, products, ",[174,1839,512],{"class":183},[174,1841,757],{"class":206},[10,1843,1844],{},"Running that before the merge rather than inspecting NaN afterwards means the message names the\nproblem in the reference data rather than describing its symptom in the output. On a scheduled job it\nis also the difference between an alert that says \"the product catalogue is stale\" and a report that\nquietly shows blanks in the description column.",[160,1846,1848],{"id":1847},"common-pitfalls","Common pitfalls",[1850,1851,1852,1868],"table",{},[1853,1854,1855],"thead",{},[1856,1857,1858,1862,1865],"tr",{},[1859,1860,1861],"th",{},"Symptom",[1859,1863,1864],{},"Cause",[1859,1866,1867],{},"Fix",[1869,1870,1871,1885,1900,1914,1927,1947],"tbody",{},[1856,1872,1873,1877,1880],{},[1874,1875,1876],"td",{},"Row count grows after a merge",[1874,1878,1879],{},"Duplicate keys in the lookup table",[1874,1881,1882,1884],{},[14,1883,773],{},", and de-duplicate the lookup",[1856,1886,1887,1890,1893],{},[1874,1888,1889],{},"Everything is NaN",[1874,1891,1892],{},"Key dtypes differ — text in one frame, integer in the other",[1874,1894,1895,1896,1899],{},"Read both with the same ",[14,1897,1898],{},"dtype",", or cast before merging",[1856,1901,1902,1905,1908],{},[1874,1903,1904],{},"Some keys match, most do not",[1874,1906,1907],{},"Whitespace or case differences",[1874,1909,1910,1913],{},[14,1911,1912],{},".str.strip().str.casefold()"," on both sides",[1856,1915,1916,1921,1924],{},[1874,1917,1918,1920],{},[14,1919,30],{}," gives wrong bands",[1874,1922,1923],{},"One or both frames unsorted",[1874,1925,1926],{},"Sort both on the join key first",[1856,1928,1929,1938,1941],{},[1874,1930,1931,1932,1165,1935],{},"Columns come back as ",[14,1933,1934],{},"Price_x",[14,1936,1937],{},"Price_y",[1874,1939,1940],{},"Both frames have a column of that name",[1874,1942,1943,1944],{},"Select the columns you need before merging, or pass ",[14,1945,1946],{},"suffixes",[1856,1948,1949,1954,1957],{},[1874,1950,1951,1953],{},[14,1952,20],{}," returns all NaN",[1874,1955,1956],{},"The Series index is not the key column",[1874,1958,1959,1962],{},[14,1960,1961],{},"set_index(key)[value]"," before mapping",[160,1964,1966],{"id":1965},"performance-and-scale","Performance and scale",[33,1968,42,1974,42,1977,42,1980,42,1984,42,1989,42,1998,42,2007,42,2013,42,2016,42,2019,42,2022,42,2027,42,2031,42,2033,42,2036,42,2039,42,2043],{"viewBox":1969,"role":36,"ariaLabelledBy":1970,"xmlns":40,"style":1973},"0 0 720 196",[1971,1972],"im-cost-t","im-cost-d","width:100%;max-width:720px;height:auto;display:block;margin:1.5rem auto;font-family:Inter,ui-sans-serif,system-ui,sans-serif",[44,1975,1976],{"id":1971},"Resolving 50,000 lookups against a 5,000-row table",[48,1978,1979],{"id":1972},"Excel's exact MATCH scans the lookup range for each row, while pandas builds a hash index of the table once and resolves every key in constant time afterwards.",[52,1981],{"x":54,"y":54,"width":1982,"height":1983,"fill":57},"720","196",[69,1985,1988],{"x":60,"y":1986,"style":1987},"56","font-size:12px;font-weight:600;fill:var(--text,#172033);text-anchor:start","MATCH, exact",[52,1990],{"x":1991,"y":1992,"width":1993,"height":1994,"rx":1995,"fill":1996,"stroke":1997},"200","40","378.5","26","6","#e7ebef","var(--line,#cdd5e6)",[52,1999],{"x":2000,"y":2001,"width":2002,"height":2003,"rx":2004,"fill":2005,"stroke":2006},"201","41","376.5","24","5","#fee8f2","var(--accent,#d81b73)",[69,2008,2012],{"x":2009,"y":2010,"style":2011},"590.5","58","font-size:12px;font-weight:700;fill:var(--accent,#d81b73);text-anchor:start","scan per row",[69,2014,16],{"x":60,"y":2015,"style":1987},"100",[52,2017],{"x":1991,"y":2018,"width":1993,"height":1994,"rx":1995,"fill":1996,"stroke":1997},"84",[52,2020],{"x":2000,"y":2021,"width":61,"height":2003,"rx":2004,"fill":65,"stroke":66},"85",[69,2023,2026],{"x":2009,"y":2024,"style":2025},"102","font-size:12px;font-weight:700;fill:var(--teal-ink,#0b6157);text-anchor:start","hash built once",[69,2028,2030],{"x":60,"y":2029,"style":1987},"144","map from a Series",[52,2032],{"x":1991,"y":127,"width":1993,"height":1994,"rx":1995,"fill":1996,"stroke":1997},[52,2034],{"x":2000,"y":2035,"width":61,"height":2003,"rx":2004,"fill":65,"stroke":66},"129",[69,2037,2026],{"x":2009,"y":2038,"style":2025},"146",[69,2040,2042],{"x":60,"y":60,"style":2041},"font-size:11.5px;font-weight:600;fill:var(--muted,#5b6780);text-anchor:start","relative cost",[69,2044,2047],{"x":2045,"y":2046,"style":157},"360.0","186","the lookup table is indexed once, not once per row",[10,2049,2050],{},"The performance story is lopsided. Excel's exact MATCH scans the lookup range for every row, so a\n50,000-row sheet against a 5,000-row table performs 250 million comparisons in the worst case.\npandas builds a hash index of the lookup table once and then resolves each key in constant time.",[165,2052,2054],{"className":193,"code":2053,"language":195,"meta":170,"style":170},"import time\n\nstart = time.perf_counter()\nresult = orders.merge(products, on=\"SKU\", how=\"left\")\nprint(f\"merge: {time.perf_counter() - start:.4f}s for {len(result):,} rows\")\n",[14,2055,2056,2063,2067,2077,2103],{"__ignoreMap":170},[174,2057,2058,2060],{"class":75,"line":176},[174,2059,203],{"class":202},[174,2061,2062],{"class":206}," time\n",[174,2064,2065],{"class":75,"line":216},[174,2066,220],{"emptyLinePlaceholder":219},[174,2068,2069,2072,2074],{"class":75,"line":223},[174,2070,2071],{"class":206},"start ",[174,2073,229],{"class":202},[174,2075,2076],{"class":206}," time.perf_counter()\n",[174,2078,2079,2082,2084,2087,2089,2091,2093,2095,2097,2099,2101],{"class":75,"line":235},[174,2080,2081],{"class":206},"result ",[174,2083,229],{"class":202},[174,2085,2086],{"class":206}," orders.merge(products, ",[174,2088,1099],{"class":720},[174,2090,229],{"class":202},[174,2092,512],{"class":183},[174,2094,248],{"class":206},[174,2096,1116],{"class":720},[174,2098,229],{"class":202},[174,2100,738],{"class":183},[174,2102,757],{"class":206},[174,2104,2105,2107,2109,2111,2114,2116,2119,2121,2124,2127,2129,2132,2134,2136,2139,2142,2144,2147],{"class":75,"line":272},[174,2106,577],{"class":244},[174,2108,835],{"class":206},[174,2110,838],{"class":202},[174,2112,2113],{"class":183},"\"merge: ",[174,2115,845],{"class":844},[174,2117,2118],{"class":206},"time.perf_counter() ",[174,2120,1721],{"class":202},[174,2122,2123],{"class":206}," start",[174,2125,2126],{"class":202},":.4f",[174,2128,854],{"class":844},[174,2130,2131],{"class":183},"s for ",[174,2133,845],{"class":844},[174,2135,848],{"class":244},[174,2137,2138],{"class":206},"(result)",[174,2140,2141],{"class":202},":,",[174,2143,854],{"class":844},[174,2145,2146],{"class":183}," rows\"",[174,2148,757],{"class":206},[10,2150,2151,2152,2155],{},"Two habits keep it fast on large data. Select only the columns you need from the lookup table before\nmerging, so the join does not carry twenty unused columns through the result. And convert repeated\nstring keys to ",[14,2153,2154],{},"category"," dtype when both sides share the same categories — the join then compares\ninteger codes.",[10,2157,2158,2159,2161],{},"The one case that is genuinely slower in pandas is ",[14,2160,30],{}," on unsorted data, because the sort\ndominates. If the same lookup runs repeatedly, sort the reference table once and keep it sorted.",[160,2163,2165],{"id":2164},"conclusion","Conclusion",[10,2167,2168,2169,2171,2172,2174,2175,2177,2178,2180],{},"INDEX\u002FMATCH becomes ",[14,2170,20],{}," for a single column and ",[14,2173,16],{}," for several, and the leftward-lookup\nproblem that motivated the formula pair stops existing. Use ",[14,2176,773],{}," so a duplicated key\nraises instead of inflating the report, count the unmatched rows rather than hiding them behind a\ndefault, pass a list for composite keys, and reach for ",[14,2179,30],{}," when the match is approximate.",[160,2182,2184],{"id":2183},"frequently-asked-questions","Frequently asked questions",[10,2186,2187,2191],{},[2188,2189,2190],"strong",{},"When should I use map instead of merge?","\nUse map when you want one column added from a lookup table keyed by a single column — it is shorter and cannot duplicate rows. Use merge when you need several columns, a composite key, or control over the join type.",[10,2193,2194,2197],{},[2188,2195,2196],{},"How do I reproduce MATCH on its own?","\nMATCH returns a position, which is rarely what you want in pandas. If you genuinely need the row number, use df.index.get_indexer or a reset_index followed by a merge; usually the position was only ever a means to fetch a value.",[10,2199,2200,2203],{},[2188,2201,2202],{},"What replaces an approximate MATCH with match type 1?","\npd.merge_asof, which joins each row to the nearest preceding key. It is the right tool for banded lookups — rate tables, tax bands, tier thresholds — and it requires both frames sorted on the key.",[10,2205,2206,2209,2210,2213],{},[2188,2207,2208],{},"Why did my row count go up after a merge?","\nThe lookup table has more than one row per key, so each source row matched several. Check with other",[174,2211,2212],{},"'Key'",".duplicated().any() before merging, and pass validate='m:1' to make pandas raise instead of silently multiplying rows.",[10,2215,2216,2219],{},[2188,2217,2218],{},"How do I handle a lookup that finds nothing?","\nA left merge leaves NaN where Excel would show #N\u002FA. Count them straight afterwards rather than filling them — an unmatched key usually means a data problem worth reporting, not a blank worth hiding.",[160,2221,2223],{"id":2222},"related","Related",[2225,2226,2227,2234,2239,2246,2253],"ul",{},[2228,2229,2230,2231,2233],"li",{},"Up one level: ",[23,2232,26],{"href":25}," — the wider function map.",[2228,2235,2236,2238],{},[23,2237,778],{"href":777}," — join semantics and the row-multiplication trap in depth.",[2228,2240,2241,2245],{},[23,2242,2244],{"href":2243},"\u002Fadvanced-data-transformation-and-cleaning\u002Fmerging-and-joining-excel-dataframes\u002Fmerge-two-excel-files-on-common-column-python\u002F","Merge Two Excel Files on a Common Column in Python"," — the same operation across two workbooks.",[2228,2247,2248,2252],{},[23,2249,2251],{"href":2250},"\u002Fadvanced-data-transformation-and-cleaning\u002Fmerging-and-joining-excel-dataframes\u002Ffind-rows-in-one-excel-file-missing-from-another\u002F","Find Rows in One Excel File Missing From Another"," — reporting the keys that did not match.",[2228,2254,2255,2259],{},[23,2256,2258],{"href":2257},"\u002Fadvanced-data-transformation-and-cleaning\u002Fexcel-formula-equivalents-in-pandas\u002Fsumif-and-sumifs-equivalent-in-pandas\u002F","SUMIF and SUMIFS Equivalent in pandas"," — aggregating once the lookup has enriched the rows.",[2261,2262,2263],"style",{},"html pre.shiki code .sMTad, html code.shiki .sMTad{--shiki-default:#6F42C1;--shiki-dark:#FFB757}html pre.shiki code .srMev, html code.shiki .srMev{--shiki-default:#032F62;--shiki-dark:#ADDCFF}html .default .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .dark .shiki span {color: var(--shiki-dark);background: var(--shiki-dark-bg);font-style: var(--shiki-dark-font-style);font-weight: var(--shiki-dark-font-weight);text-decoration: var(--shiki-dark-text-decoration);}html.dark .shiki span {color: var(--shiki-dark);background: var(--shiki-dark-bg);font-style: var(--shiki-dark-font-style);font-weight: var(--shiki-dark-font-weight);text-decoration: var(--shiki-dark-text-decoration);}html pre.shiki code .s-kum, html code.shiki .s-kum{--shiki-default:#D73A49;--shiki-dark:#FF9492}html pre.shiki code .skGVy, html code.shiki .skGVy{--shiki-default:#24292E;--shiki-dark:#F0F3F6}html pre.shiki code .sP0c6, html code.shiki .sP0c6{--shiki-default:#005CC5;--shiki-dark:#91CBFF}html pre.shiki code .s-wDw, html code.shiki .s-wDw{--shiki-default:#6A737D;--shiki-dark:#BDC4CC}html pre.shiki code .sa561, html code.shiki .sa561{--shiki-default:#E36209;--shiki-dark:#FFB757}html pre.shiki code .sSjpA, html code.shiki .sSjpA{--shiki-default:#005CC5;--shiki-dark:#FF9492}html pre.shiki code .s_Opv, html code.shiki .s_Opv{--shiki-default:#6F42C1;--shiki-dark:#DBB7FF}",{"title":170,"searchDepth":216,"depth":216,"links":2265},[2266,2267,2268,2269,2270,2271,2272,2273,2274,2275,2276,2277,2278],{"id":162,"depth":216,"text":163},{"id":481,"depth":216,"text":482},{"id":664,"depth":216,"text":665},{"id":782,"depth":216,"text":783},{"id":912,"depth":216,"text":913},{"id":1248,"depth":216,"text":1249},{"id":1427,"depth":216,"text":1428},{"id":1666,"depth":216,"text":1667},{"id":1847,"depth":216,"text":1848},{"id":1965,"depth":216,"text":1966},{"id":2164,"depth":216,"text":2165},{"id":2183,"depth":216,"text":2184},{"id":2222,"depth":216,"text":2223},"2026-09-04","INDEX\u002FMATCH becomes map for one column and merge for several. Handle composite keys without a helper column, report unmatched keys, and use merge_asof for banded lookups.","md",[2283,2285,2287,2289,2291],{"q":2190,"a":2284},"Use map when you want one column added from a lookup table keyed by a single column — it is shorter and cannot duplicate rows. Use merge when you need several columns, a composite key, or control over the join type.",{"q":2196,"a":2286},"MATCH returns a position, which is rarely what you want in pandas. If you genuinely need the row number, use df.index.get_indexer or a reset_index followed by a merge; usually the position was only ever a means to fetch a value.",{"q":2202,"a":2288},"pd.merge_asof, which joins each row to the nearest preceding key. It is the right tool for banded lookups — rate tables, tax bands, tier thresholds — and it requires both frames sorted on the key.",{"q":2208,"a":2290},"The lookup table has more than one row per key, so each source row matched several. Check with other['Key'].duplicated().any() before merging, and pass validate='m:1' to make pandas raise instead of silently multiplying rows.",{"q":2218,"a":2292},"A left merge leaves NaN where Excel would show #N\u002FA. Count them straight afterwards rather than filling them — an unmatched key usually means a data problem worth reporting, not a blank worth hiding.",{"breadcrumb":2294},[2295,2298,2301],{"name":2296,"item":2297},"Home","\u002F",{"name":2299,"item":2300},"Advanced Data Transformation and Cleaning","\u002Fadvanced-data-transformation-and-cleaning\u002F",{"name":26,"item":25},"\u002Fadvanced-data-transformation-and-cleaning\u002Fexcel-formula-equivalents-in-pandas\u002Findex-match-equivalent-in-pandas",{"title":5,"description":2304},"Replace INDEX\u002FMATCH with pandas map and merge: single and multi-column lookups, composite keys, validate='m:1' against row multiplication, and merge_asof for approximate matches.","index-match-equivalent-in-pandas","advanced-data-transformation-and-cleaning\u002Fexcel-formula-equivalents-in-pandas\u002Findex-match-equivalent-in-pandas\u002Findex","how-to","6KI8QvoKTD9H-zBLMA5E0Web4lTtYY1swX0y1182enE",[2310,2314],{"title":2311,"path":2312,"stem":2313,"children":-1},"Excel Text Functions LEFT, RIGHT, MID and CONCAT in pandas","\u002Fadvanced-data-transformation-and-cleaning\u002Fexcel-formula-equivalents-in-pandas\u002Fexcel-text-functions-left-right-mid-and-concat-in-pandas","advanced-data-transformation-and-cleaning\u002Fexcel-formula-equivalents-in-pandas\u002Fexcel-text-functions-left-right-mid-and-concat-in-pandas\u002Findex",{"title":2315,"path":2316,"stem":2317,"children":-1},"RANK and PERCENTILE Formulas in pandas","\u002Fadvanced-data-transformation-and-cleaning\u002Fexcel-formula-equivalents-in-pandas\u002Frank-and-percentile-formulas-in-pandas","advanced-data-transformation-and-cleaning\u002Fexcel-formula-equivalents-in-pandas\u002Frank-and-percentile-formulas-in-pandas\u002Findex",1788710154434]