[{"data":1,"prerenderedAt":2414},["ShallowReactive",2],{"doc:\u002Fadvanced-data-transformation-and-cleaning\u002Fvalidating-excel-data-with-python\u002Ffind-duplicate-rows-in-excel-with-python":3,"surround:\u002Fadvanced-data-transformation-and-cleaning\u002Fvalidating-excel-data-with-python\u002Ffind-duplicate-rows-in-excel-with-python":2407},{"id":4,"title":5,"body":6,"dateModified":2376,"datePublished":2376,"description":2377,"extension":2378,"faq":2379,"meta":2389,"navigation":235,"path":2400,"seo":2401,"slug":2403,"stem":2404,"type":2405,"__hash__":2406},"docs\u002Fadvanced-data-transformation-and-cleaning\u002Fvalidating-excel-data-with-python\u002Ffind-duplicate-rows-in-excel-with-python\u002Findex.md","Find Duplicate Rows in Excel with Python",{"type":7,"value":8,"toc":2360},"minimark",[9,19,169,174,203,207,537,541,551,687,693,697,700,976,983,987,990,1266,1284,1395,1399,1402,1714,1731,1735,1738,1893,1905,1909,1912,1986,1989,1993,2096,2100,2103,2227,2244,2248,2257,2261,2271,2275,2282,2291,2303,2309,2313,2316,2325,2328,2356],[10,11,12,13,18],"p",{},"Duplicates are the validation problem Excel cannot solve for you: its own validation rules see one cell at a time, so nothing in the workbook can notice that order 2001 appears twice. pandas does it in a line — and then the interesting work starts, because \"duplicate\" turns out to mean at least three different things. This guide, part of ",[14,15,17],"a",{"href":16},"\u002Fadvanced-data-transformation-and-cleaning\u002Fvalidating-excel-data-with-python\u002F","Validating Excel Data with Python",", covers all of them.",[20,21,29,30,29,34,29,38,29,45,29,55,29,62,29,70,29,75,29,79,29,82,29,87,29,92,29,96,29,101,29,105,29,110,29,113,29,115,29,117,29,121,29,124,29,127,29,130,29,134,29,139,29,144,29,147,29,150,29,152,29,156,29,159,29,162,29,165],"svg",{"viewBox":22,"role":23,"ariaLabelledBy":24,"xmlns":27,"style":28},"0 0 740 256","img",[25,26],"dup-kinds-t","dup-kinds-d","http:\u002F\u002Fwww.w3.org\u002F2000\u002Fsvg","width:100%;max-width:740px;height:auto;display:block;margin:1.5rem auto;font-family:Inter,ui-sans-serif,system-ui,sans-serif","\n  ",[31,32,33],"title",{"id":25},"Three kinds of duplicate, and what each one means",[35,36,37],"desc",{"id":26},"An exact duplicate repeats every column and is usually a double import. A duplicate key repeats the identifier while other values differ, meaning two records disagree. A near duplicate differs only by whitespace or case and is a normalisation problem.",[39,40],"rect",{"x":41,"y":41,"width":42,"height":43,"fill":44},"0","740","256","#ffffff",[39,46],{"x":47,"y":48,"width":49,"height":50,"rx":51,"fill":52,"stroke":53,"style":54},"16","26","228","212","14","#f0f4ff","var(--brand,#5b5cf0)","stroke-width:2px",[56,57,61],"text",{"x":58,"y":59,"style":60},"130","52","font-size:13px;font-weight:700;fill:var(--brand-strong,#4338ca);text-anchor:middle","exact duplicate",[39,63],{"x":64,"y":65,"width":66,"height":67,"rx":68,"fill":44,"stroke":69},"36","66","188","28","6","var(--line,#cdd5e6)",[56,71,74],{"x":58,"y":72,"style":73},"85","font-size:11px;fill:var(--text,#172033);text-anchor:middle","2001 · North · 4 · 19.99",[39,76],{"x":64,"y":77,"width":66,"height":67,"rx":68,"fill":78,"stroke":69},"98","#ebebfd",[56,80,74],{"x":58,"y":81,"style":73},"117",[56,83,86],{"x":58,"y":84,"style":85},"152","font-size:11.5px;fill:var(--muted,#5b6780);text-anchor:middle","every column matches",[56,88,91],{"x":58,"y":89,"style":90},"178","font-size:11.5px;font-weight:700;fill:var(--text,#172033);text-anchor:middle","cause: double import,",[56,93,95],{"x":58,"y":94,"style":90},"196","copy-paste",[56,97,100],{"x":58,"y":98,"style":99},"222","font-size:11.5px;fill:var(--teal,#0f9488);text-anchor:middle","usually safe to drop",[39,102],{"x":43,"y":48,"width":49,"height":50,"rx":51,"fill":103,"stroke":104,"style":54},"#fce9e9","var(--danger,#dc2626)",[56,106,109],{"x":107,"y":59,"style":108},"370","font-size:13px;font-weight:700;fill:var(--danger,#dc2626);text-anchor:middle","duplicate key",[39,111],{"x":112,"y":65,"width":66,"height":67,"rx":68,"fill":44,"stroke":69},"276",[56,114,74],{"x":107,"y":72,"style":73},[39,116],{"x":112,"y":77,"width":66,"height":67,"rx":68,"fill":44,"stroke":104},[56,118,120],{"x":107,"y":81,"style":119},"font-size:11px;font-weight:700;fill:var(--danger,#dc2626);text-anchor:middle","2001 · West · 3 · 8.75",[56,122,123],{"x":107,"y":84,"style":85},"same id, different data",[56,125,126],{"x":107,"y":89,"style":90},"cause: two systems",[56,128,129],{"x":107,"y":94,"style":90},"disagree",[56,131,133],{"x":107,"y":98,"style":132},"font-size:11.5px;fill:var(--danger,#dc2626);text-anchor:middle","never drop silently",[39,135],{"x":136,"y":48,"width":49,"height":50,"rx":51,"fill":137,"stroke":138,"style":54},"496","#fdefd8","var(--gold,#b4740a)",[56,140,143],{"x":141,"y":59,"style":142},"610","font-size:13px;font-weight:700;fill:var(--gold-ink,#7a4e06);text-anchor:middle","near duplicate",[39,145],{"x":146,"y":65,"width":66,"height":67,"rx":68,"fill":44,"stroke":69},"516",[56,148,149],{"x":141,"y":72,"style":73},"\"North\" · A-100",[39,151],{"x":146,"y":77,"width":66,"height":67,"rx":68,"fill":44,"stroke":138},[56,153,155],{"x":141,"y":81,"style":154},"font-size:11px;font-weight:700;fill:var(--gold-ink,#7a4e06);text-anchor:middle","\"north \" · a-100",[56,157,158],{"x":141,"y":84,"style":85},"differs by case or spaces",[56,160,161],{"x":141,"y":89,"style":90},"cause: typing, not",[56,163,164],{"x":141,"y":94,"style":90},"duplication",[56,166,168],{"x":141,"y":98,"style":167},"font-size:11.5px;fill:var(--gold,#b4740a);text-anchor:middle","normalise, then re-check",[170,171,173],"h2",{"id":172},"prerequisites","Prerequisites",[175,176,181],"pre",{"className":177,"code":178,"language":179,"meta":180,"style":180},"language-bash shiki shiki-themes github-light github-dark-high-contrast","pip install pandas openpyxl\n","bash","",[182,183,184],"code",{"__ignoreMap":180},[185,186,189,193,197,200],"span",{"class":187,"line":188},"line",1,[185,190,192],{"class":191},"sMTad","pip",[185,194,196],{"class":195},"srMev"," install",[185,198,199],{"class":195}," pandas",[185,201,202],{"class":195}," openpyxl\n",[170,204,206],{"id":205},"a-sample-with-all-three-kinds","A sample with all three kinds",[175,208,212],{"className":209,"code":210,"language":211,"meta":180,"style":180},"language-python shiki shiki-themes github-light github-dark-high-contrast","import pandas as pd\n\npd.DataFrame([\n    {\"Order_ID\": 2001, \"SKU\": \"A-100\", \"Region\": \"North\", \"Quantity\": 4},\n    {\"Order_ID\": 2002, \"SKU\": \"B-200\", \"Region\": \"South\", \"Quantity\": 2},\n    {\"Order_ID\": 2001, \"SKU\": \"A-100\", \"Region\": \"North\", \"Quantity\": 4},   # exact copy\n    {\"Order_ID\": 2003, \"SKU\": \"C-300\", \"Region\": \"West\", \"Quantity\": 7},\n    {\"Order_ID\": 2003, \"SKU\": \"C-300\", \"Region\": \"West\", \"Quantity\": 9},    # key clash\n    {\"Order_ID\": 2004, \"SKU\": \"a-100 \", \"Region\": \"north\", \"Quantity\": 4},  # near duplicate\n]).to_excel(\"orders.xlsx\", index=False, sheet_name=\"Orders\")\n","python",[182,213,214,230,237,243,293,334,376,417,459,503],{"__ignoreMap":180},[185,215,216,220,224,227],{"class":187,"line":188},[185,217,219],{"class":218},"s-kum","import",[185,221,223],{"class":222},"skGVy"," pandas ",[185,225,226],{"class":218},"as",[185,228,229],{"class":222}," pd\n",[185,231,233],{"class":187,"line":232},2,[185,234,236],{"emptyLinePlaceholder":235},true,"\n",[185,238,240],{"class":187,"line":239},3,[185,241,242],{"class":222},"pd.DataFrame([\n",[185,244,246,249,252,255,259,262,265,267,270,272,275,277,280,282,285,287,290],{"class":187,"line":245},4,[185,247,248],{"class":222},"    {",[185,250,251],{"class":195},"\"Order_ID\"",[185,253,254],{"class":222},": ",[185,256,258],{"class":257},"sP0c6","2001",[185,260,261],{"class":222},", ",[185,263,264],{"class":195},"\"SKU\"",[185,266,254],{"class":222},[185,268,269],{"class":195},"\"A-100\"",[185,271,261],{"class":222},[185,273,274],{"class":195},"\"Region\"",[185,276,254],{"class":222},[185,278,279],{"class":195},"\"North\"",[185,281,261],{"class":222},[185,283,284],{"class":195},"\"Quantity\"",[185,286,254],{"class":222},[185,288,289],{"class":257},"4",[185,291,292],{"class":222},"},\n",[185,294,296,298,300,302,305,307,309,311,314,316,318,320,323,325,327,329,332],{"class":187,"line":295},5,[185,297,248],{"class":222},[185,299,251],{"class":195},[185,301,254],{"class":222},[185,303,304],{"class":257},"2002",[185,306,261],{"class":222},[185,308,264],{"class":195},[185,310,254],{"class":222},[185,312,313],{"class":195},"\"B-200\"",[185,315,261],{"class":222},[185,317,274],{"class":195},[185,319,254],{"class":222},[185,321,322],{"class":195},"\"South\"",[185,324,261],{"class":222},[185,326,284],{"class":195},[185,328,254],{"class":222},[185,330,331],{"class":257},"2",[185,333,292],{"class":222},[185,335,337,339,341,343,345,347,349,351,353,355,357,359,361,363,365,367,369,372],{"class":187,"line":336},6,[185,338,248],{"class":222},[185,340,251],{"class":195},[185,342,254],{"class":222},[185,344,258],{"class":257},[185,346,261],{"class":222},[185,348,264],{"class":195},[185,350,254],{"class":222},[185,352,269],{"class":195},[185,354,261],{"class":222},[185,356,274],{"class":195},[185,358,254],{"class":222},[185,360,279],{"class":195},[185,362,261],{"class":222},[185,364,284],{"class":195},[185,366,254],{"class":222},[185,368,289],{"class":257},[185,370,371],{"class":222},"},   ",[185,373,375],{"class":374},"s-wDw","# exact copy\n",[185,377,379,381,383,385,388,390,392,394,397,399,401,403,406,408,410,412,415],{"class":187,"line":378},7,[185,380,248],{"class":222},[185,382,251],{"class":195},[185,384,254],{"class":222},[185,386,387],{"class":257},"2003",[185,389,261],{"class":222},[185,391,264],{"class":195},[185,393,254],{"class":222},[185,395,396],{"class":195},"\"C-300\"",[185,398,261],{"class":222},[185,400,274],{"class":195},[185,402,254],{"class":222},[185,404,405],{"class":195},"\"West\"",[185,407,261],{"class":222},[185,409,284],{"class":195},[185,411,254],{"class":222},[185,413,414],{"class":257},"7",[185,416,292],{"class":222},[185,418,420,422,424,426,428,430,432,434,436,438,440,442,444,446,448,450,453,456],{"class":187,"line":419},8,[185,421,248],{"class":222},[185,423,251],{"class":195},[185,425,254],{"class":222},[185,427,387],{"class":257},[185,429,261],{"class":222},[185,431,264],{"class":195},[185,433,254],{"class":222},[185,435,396],{"class":195},[185,437,261],{"class":222},[185,439,274],{"class":195},[185,441,254],{"class":222},[185,443,405],{"class":195},[185,445,261],{"class":222},[185,447,284],{"class":195},[185,449,254],{"class":222},[185,451,452],{"class":257},"9",[185,454,455],{"class":222},"},    ",[185,457,458],{"class":374},"# key clash\n",[185,460,462,464,466,468,471,473,475,477,480,482,484,486,489,491,493,495,497,500],{"class":187,"line":461},9,[185,463,248],{"class":222},[185,465,251],{"class":195},[185,467,254],{"class":222},[185,469,470],{"class":257},"2004",[185,472,261],{"class":222},[185,474,264],{"class":195},[185,476,254],{"class":222},[185,478,479],{"class":195},"\"a-100 \"",[185,481,261],{"class":222},[185,483,274],{"class":195},[185,485,254],{"class":222},[185,487,488],{"class":195},"\"north\"",[185,490,261],{"class":222},[185,492,284],{"class":195},[185,494,254],{"class":222},[185,496,289],{"class":257},[185,498,499],{"class":222},"},  ",[185,501,502],{"class":374},"# near duplicate\n",[185,504,506,509,512,514,518,521,524,526,529,531,534],{"class":187,"line":505},10,[185,507,508],{"class":222},"]).to_excel(",[185,510,511],{"class":195},"\"orders.xlsx\"",[185,513,261],{"class":222},[185,515,517],{"class":516},"sa561","index",[185,519,520],{"class":218},"=",[185,522,523],{"class":257},"False",[185,525,261],{"class":222},[185,527,528],{"class":516},"sheet_name",[185,530,520],{"class":218},[185,532,533],{"class":195},"\"Orders\"",[185,535,536],{"class":222},")\n",[170,538,540],{"id":539},"exact-duplicates","Exact duplicates",[10,542,543,546,547,550],{},[182,544,545],{},"duplicated()"," compares whole rows. ",[182,548,549],{},"keep=False"," is the important argument — the default marks only the second and later copies, which is what you want for dropping and not what you want for reporting:",[175,552,554],{"className":209,"code":553,"language":211,"meta":180,"style":180},"import pandas as pd\n\ndf = pd.read_excel(\"orders.xlsx\", sheet_name=\"Orders\")\n\nrepeats_only = df[df.duplicated()]                 # the copies\nevery_copy = df[df.duplicated(keep=False)]         # copies and originals\n\nprint(f\"{len(repeats_only)} redundant row(s), {len(every_copy)} row(s) involved\")\nprint(every_copy)\n",[182,555,556,566,570,592,596,609,632,636,680],{"__ignoreMap":180},[185,557,558,560,562,564],{"class":187,"line":188},[185,559,219],{"class":218},[185,561,223],{"class":222},[185,563,226],{"class":218},[185,565,229],{"class":222},[185,567,568],{"class":187,"line":232},[185,569,236],{"emptyLinePlaceholder":235},[185,571,572,575,577,580,582,584,586,588,590],{"class":187,"line":239},[185,573,574],{"class":222},"df ",[185,576,520],{"class":218},[185,578,579],{"class":222}," pd.read_excel(",[185,581,511],{"class":195},[185,583,261],{"class":222},[185,585,528],{"class":516},[185,587,520],{"class":218},[185,589,533],{"class":195},[185,591,536],{"class":222},[185,593,594],{"class":187,"line":245},[185,595,236],{"emptyLinePlaceholder":235},[185,597,598,601,603,606],{"class":187,"line":295},[185,599,600],{"class":222},"repeats_only ",[185,602,520],{"class":218},[185,604,605],{"class":222}," df[df.duplicated()]                 ",[185,607,608],{"class":374},"# the copies\n",[185,610,611,614,616,619,622,624,626,629],{"class":187,"line":336},[185,612,613],{"class":222},"every_copy ",[185,615,520],{"class":218},[185,617,618],{"class":222}," df[df.duplicated(",[185,620,621],{"class":516},"keep",[185,623,520],{"class":218},[185,625,523],{"class":257},[185,627,628],{"class":222},")]         ",[185,630,631],{"class":374},"# copies and originals\n",[185,633,634],{"class":187,"line":378},[185,635,236],{"emptyLinePlaceholder":235},[185,637,638,641,644,647,650,654,657,660,663,666,668,670,673,675,678],{"class":187,"line":419},[185,639,640],{"class":257},"print",[185,642,643],{"class":222},"(",[185,645,646],{"class":218},"f",[185,648,649],{"class":195},"\"",[185,651,653],{"class":652},"sSjpA","{",[185,655,656],{"class":257},"len",[185,658,659],{"class":222},"(repeats_only)",[185,661,662],{"class":652},"}",[185,664,665],{"class":195}," redundant row(s), ",[185,667,653],{"class":652},[185,669,656],{"class":257},[185,671,672],{"class":222},"(every_copy)",[185,674,662],{"class":652},[185,676,677],{"class":195}," row(s) involved\"",[185,679,536],{"class":222},[185,681,682,684],{"class":187,"line":461},[185,683,640],{"class":257},[185,685,686],{"class":222},"(every_copy)\n",[10,688,689,690,692],{},"Reporting with ",[182,691,549],{}," lets a reader see both rows and confirm they really are identical. Reporting with the default shows one row of a pair, which invariably prompts the question \"duplicate of what?\".",[170,694,696],{"id":695},"duplicate-keys-the-important-case","Duplicate keys — the important case",[10,698,699],{},"An identifier that repeats while other values differ means two records claim to describe the same thing. This is a genuine data conflict, and it is the check most reporting pipelines are missing:",[175,701,703],{"className":209,"code":702,"language":211,"meta":180,"style":180},"import pandas as pd\n\nKEY = [\"Order_ID\"]\n\ndef key_conflicts(frame, key=KEY):\n    dupes = frame[frame.duplicated(subset=key, keep=False)].copy()\n    if dupes.empty:\n        return dupes.assign(Conflict=None)\n\n    other = [c for c in frame.columns if c not in key]\n    # A conflict is a repeated key whose remaining columns are not all identical\n    grouped = dupes.groupby(key)[other].nunique()\n    conflicted = grouped[(grouped > 1).any(axis=1)].index\n\n    dupes[\"Conflict\"] = dupes[key[0]].isin(conflicted)\n    return dupes.sort_values(key)\n\nconflicts = key_conflicts(df)\nprint(conflicts[[\"Order_ID\", \"Region\", \"Quantity\", \"Conflict\"]])\n",[182,704,705,715,719,735,739,758,785,793,811,815,851,857,868,899,904,926,935,940,951],{"__ignoreMap":180},[185,706,707,709,711,713],{"class":187,"line":188},[185,708,219],{"class":218},[185,710,223],{"class":222},[185,712,226],{"class":218},[185,714,229],{"class":222},[185,716,717],{"class":187,"line":232},[185,718,236],{"emptyLinePlaceholder":235},[185,720,721,724,727,730,732],{"class":187,"line":239},[185,722,723],{"class":257},"KEY",[185,725,726],{"class":218}," =",[185,728,729],{"class":222}," [",[185,731,251],{"class":195},[185,733,734],{"class":222},"]\n",[185,736,737],{"class":187,"line":245},[185,738,236],{"emptyLinePlaceholder":235},[185,740,741,744,748,751,753,755],{"class":187,"line":295},[185,742,743],{"class":218},"def",[185,745,747],{"class":746},"s_Opv"," key_conflicts",[185,749,750],{"class":222},"(frame, key",[185,752,520],{"class":218},[185,754,723],{"class":257},[185,756,757],{"class":222},"):\n",[185,759,760,763,765,768,771,773,776,778,780,782],{"class":187,"line":336},[185,761,762],{"class":222},"    dupes ",[185,764,520],{"class":218},[185,766,767],{"class":222}," frame[frame.duplicated(",[185,769,770],{"class":516},"subset",[185,772,520],{"class":218},[185,774,775],{"class":222},"key, ",[185,777,621],{"class":516},[185,779,520],{"class":218},[185,781,523],{"class":257},[185,783,784],{"class":222},")].copy()\n",[185,786,787,790],{"class":187,"line":378},[185,788,789],{"class":218},"    if",[185,791,792],{"class":222}," dupes.empty:\n",[185,794,795,798,801,804,806,809],{"class":187,"line":419},[185,796,797],{"class":218},"        return",[185,799,800],{"class":222}," dupes.assign(",[185,802,803],{"class":516},"Conflict",[185,805,520],{"class":218},[185,807,808],{"class":257},"None",[185,810,536],{"class":222},[185,812,813],{"class":187,"line":461},[185,814,236],{"emptyLinePlaceholder":235},[185,816,817,820,822,825,828,831,834,837,840,842,845,848],{"class":187,"line":505},[185,818,819],{"class":222},"    other ",[185,821,520],{"class":218},[185,823,824],{"class":222}," [c ",[185,826,827],{"class":218},"for",[185,829,830],{"class":222}," c ",[185,832,833],{"class":218},"in",[185,835,836],{"class":222}," frame.columns ",[185,838,839],{"class":218},"if",[185,841,830],{"class":222},[185,843,844],{"class":218},"not",[185,846,847],{"class":218}," in",[185,849,850],{"class":222}," key]\n",[185,852,854],{"class":187,"line":853},11,[185,855,856],{"class":374},"    # A conflict is a repeated key whose remaining columns are not all identical\n",[185,858,860,863,865],{"class":187,"line":859},12,[185,861,862],{"class":222},"    grouped ",[185,864,520],{"class":218},[185,866,867],{"class":222}," dupes.groupby(key)[other].nunique()\n",[185,869,871,874,876,879,882,885,888,891,893,896],{"class":187,"line":870},13,[185,872,873],{"class":222},"    conflicted ",[185,875,520],{"class":218},[185,877,878],{"class":222}," grouped[(grouped ",[185,880,881],{"class":218},">",[185,883,884],{"class":257}," 1",[185,886,887],{"class":222},").any(",[185,889,890],{"class":516},"axis",[185,892,520],{"class":218},[185,894,895],{"class":257},"1",[185,897,898],{"class":222},")].index\n",[185,900,902],{"class":187,"line":901},14,[185,903,236],{"emptyLinePlaceholder":235},[185,905,907,910,913,916,918,921,923],{"class":187,"line":906},15,[185,908,909],{"class":222},"    dupes[",[185,911,912],{"class":195},"\"Conflict\"",[185,914,915],{"class":222},"] ",[185,917,520],{"class":218},[185,919,920],{"class":222}," dupes[key[",[185,922,41],{"class":257},[185,924,925],{"class":222},"]].isin(conflicted)\n",[185,927,929,932],{"class":187,"line":928},16,[185,930,931],{"class":218},"    return",[185,933,934],{"class":222}," dupes.sort_values(key)\n",[185,936,938],{"class":187,"line":937},17,[185,939,236],{"emptyLinePlaceholder":235},[185,941,943,946,948],{"class":187,"line":942},18,[185,944,945],{"class":222},"conflicts ",[185,947,520],{"class":218},[185,949,950],{"class":222}," key_conflicts(df)\n",[185,952,954,956,959,961,963,965,967,969,971,973],{"class":187,"line":953},19,[185,955,640],{"class":257},[185,957,958],{"class":222},"(conflicts[[",[185,960,251],{"class":195},[185,962,261],{"class":222},[185,964,274],{"class":195},[185,966,261],{"class":222},[185,968,284],{"class":195},[185,970,261],{"class":222},[185,972,912],{"class":195},[185,974,975],{"class":222},"]])\n",[10,977,978,979,982],{},"The ",[182,980,981],{},"nunique"," step is what separates a harmless double import from a real disagreement: if every non-key column has a single distinct value within the group, the rows are copies. If any column has two, the two rows say different things about order 2003, and no automatic rule can decide which is right.",[170,984,986],{"id":985},"normalise-before-comparing","Normalise before comparing",[10,988,989],{},"Exact comparison misses the third kind entirely. Build a comparison key with the same normalisation you would apply anywhere else — trim, case-fold, and coerce numbers — but keep the original values for the report:",[175,991,993],{"className":209,"code":992,"language":211,"meta":180,"style":180},"import pandas as pd\n\ndef normalised_key(frame, columns):\n    key = pd.Series(\"\", index=frame.index)\n    for column in columns:\n        values = frame[column]\n        if pd.api.types.is_numeric_dtype(values):\n            part = values.round(2).astype(str)\n        else:\n            part = (values.astype(str).str.strip().str.casefold()\n                    .str.replace(r\"\\s+\", \" \", regex=True))\n        key = key + \"␟\" + part            # a separator no value will contain\n    return key\n\ndf[\"_key\"] = normalised_key(df, [\"SKU\", \"Region\", \"Quantity\"])\nnear = df[df.duplicated(subset=\"_key\", keep=False)].sort_values(\"_key\")\nprint(near[[\"Order_ID\", \"SKU\", \"Region\", \"Quantity\"]])\n",[182,994,995,1005,1009,1019,1041,1054,1064,1072,1092,1100,1114,1150,1174,1181,1185,1213,1243],{"__ignoreMap":180},[185,996,997,999,1001,1003],{"class":187,"line":188},[185,998,219],{"class":218},[185,1000,223],{"class":222},[185,1002,226],{"class":218},[185,1004,229],{"class":222},[185,1006,1007],{"class":187,"line":232},[185,1008,236],{"emptyLinePlaceholder":235},[185,1010,1011,1013,1016],{"class":187,"line":239},[185,1012,743],{"class":218},[185,1014,1015],{"class":746}," normalised_key",[185,1017,1018],{"class":222},"(frame, columns):\n",[185,1020,1021,1024,1026,1029,1032,1034,1036,1038],{"class":187,"line":245},[185,1022,1023],{"class":222},"    key ",[185,1025,520],{"class":218},[185,1027,1028],{"class":222}," pd.Series(",[185,1030,1031],{"class":195},"\"\"",[185,1033,261],{"class":222},[185,1035,517],{"class":516},[185,1037,520],{"class":218},[185,1039,1040],{"class":222},"frame.index)\n",[185,1042,1043,1046,1049,1051],{"class":187,"line":295},[185,1044,1045],{"class":218},"    for",[185,1047,1048],{"class":222}," column ",[185,1050,833],{"class":218},[185,1052,1053],{"class":222}," columns:\n",[185,1055,1056,1059,1061],{"class":187,"line":336},[185,1057,1058],{"class":222},"        values ",[185,1060,520],{"class":218},[185,1062,1063],{"class":222}," frame[column]\n",[185,1065,1066,1069],{"class":187,"line":378},[185,1067,1068],{"class":218},"        if",[185,1070,1071],{"class":222}," pd.api.types.is_numeric_dtype(values):\n",[185,1073,1074,1077,1079,1082,1084,1087,1090],{"class":187,"line":419},[185,1075,1076],{"class":222},"            part ",[185,1078,520],{"class":218},[185,1080,1081],{"class":222}," values.round(",[185,1083,331],{"class":257},[185,1085,1086],{"class":222},").astype(",[185,1088,1089],{"class":257},"str",[185,1091,536],{"class":222},[185,1093,1094,1097],{"class":187,"line":461},[185,1095,1096],{"class":218},"        else",[185,1098,1099],{"class":222},":\n",[185,1101,1102,1104,1106,1109,1111],{"class":187,"line":505},[185,1103,1076],{"class":222},[185,1105,520],{"class":218},[185,1107,1108],{"class":222}," (values.astype(",[185,1110,1089],{"class":257},[185,1112,1113],{"class":222},").str.strip().str.casefold()\n",[185,1115,1116,1119,1122,1124,1127,1130,1132,1134,1137,1139,1142,1144,1147],{"class":187,"line":853},[185,1117,1118],{"class":222},"                    .str.replace(",[185,1120,1121],{"class":218},"r",[185,1123,649],{"class":195},[185,1125,1126],{"class":257},"\\s",[185,1128,1129],{"class":218},"+",[185,1131,649],{"class":195},[185,1133,261],{"class":222},[185,1135,1136],{"class":195},"\" \"",[185,1138,261],{"class":222},[185,1140,1141],{"class":516},"regex",[185,1143,520],{"class":218},[185,1145,1146],{"class":257},"True",[185,1148,1149],{"class":222},"))\n",[185,1151,1152,1155,1157,1160,1162,1165,1168,1171],{"class":187,"line":859},[185,1153,1154],{"class":222},"        key ",[185,1156,520],{"class":218},[185,1158,1159],{"class":222}," key ",[185,1161,1129],{"class":218},[185,1163,1164],{"class":195}," \"␟\"",[185,1166,1167],{"class":218}," +",[185,1169,1170],{"class":222}," part            ",[185,1172,1173],{"class":374},"# a separator no value will contain\n",[185,1175,1176,1178],{"class":187,"line":870},[185,1177,931],{"class":218},[185,1179,1180],{"class":222}," key\n",[185,1182,1183],{"class":187,"line":901},[185,1184,236],{"emptyLinePlaceholder":235},[185,1186,1187,1190,1193,1195,1197,1200,1202,1204,1206,1208,1210],{"class":187,"line":906},[185,1188,1189],{"class":222},"df[",[185,1191,1192],{"class":195},"\"_key\"",[185,1194,915],{"class":222},[185,1196,520],{"class":218},[185,1198,1199],{"class":222}," normalised_key(df, [",[185,1201,264],{"class":195},[185,1203,261],{"class":222},[185,1205,274],{"class":195},[185,1207,261],{"class":222},[185,1209,284],{"class":195},[185,1211,1212],{"class":222},"])\n",[185,1214,1215,1218,1220,1222,1224,1226,1228,1230,1232,1234,1236,1239,1241],{"class":187,"line":928},[185,1216,1217],{"class":222},"near ",[185,1219,520],{"class":218},[185,1221,618],{"class":222},[185,1223,770],{"class":516},[185,1225,520],{"class":218},[185,1227,1192],{"class":195},[185,1229,261],{"class":222},[185,1231,621],{"class":516},[185,1233,520],{"class":218},[185,1235,523],{"class":257},[185,1237,1238],{"class":222},")].sort_values(",[185,1240,1192],{"class":195},[185,1242,536],{"class":222},[185,1244,1245,1247,1250,1252,1254,1256,1258,1260,1262,1264],{"class":187,"line":937},[185,1246,640],{"class":257},[185,1248,1249],{"class":222},"(near[[",[185,1251,251],{"class":195},[185,1253,261],{"class":222},[185,1255,264],{"class":195},[185,1257,261],{"class":222},[185,1259,274],{"class":195},[185,1261,261],{"class":222},[185,1263,284],{"class":195},[185,1265,975],{"class":222},[10,1267,1268,1271,1272,1275,1276,1279,1280,1283],{},[182,1269,1270],{},"casefold"," rather than ",[182,1273,1274],{},"lower"," is deliberate — it handles non-English text correctly, and validation data has a habit of containing a German street name eventually. Rounding numerics before comparison stops ",[182,1277,1278],{},"4.0"," and ",[182,1281,1282],{},"4.000000001"," counting as different, which happens whenever a value has been through a floating-point calculation.",[20,1285,29,1290,29,1293,29,1296,29,1299,29,1305,29,1311,29,1316,29,1319,29,1323,29,1328,29,1335,29,1340,29,1344,29,1347,29,1353,29,1359,29,1363,29,1369,29,1373,29,1379,29,1385,29,1390],{"viewBox":1286,"role":23,"ariaLabelledBy":1287,"xmlns":27,"style":28},"0 0 740 226",[1288,1289],"dup-key-t","dup-key-d",[31,1291,1292],{"id":1288},"Building a comparison key from normalised columns",[35,1294,1295],{"id":1289},"Two rows that look different — A-100 with North and quantity 4, and a-100 with a trailing space, north lowercase and quantity 4.0 — produce the same normalised key after trimming, case folding and rounding, so they are detected as near duplicates.",[39,1297],{"x":41,"y":41,"width":42,"height":1298,"fill":44},"226",[39,1300],{"x":1301,"y":1302,"width":1303,"height":65,"rx":1304,"fill":44,"stroke":69},"24","34","300","10",[56,1306,1310],{"x":1307,"y":1308,"style":1309},"174","58","font-size:11.5px;font-weight:700;fill:var(--muted,#5b6780);text-anchor:middle","row 2",[56,1312,1315],{"x":1307,"y":1313,"style":1314},"82","font-size:12.5px;fill:var(--text,#172033);text-anchor:middle;font-family:ui-monospace,Menlo,monospace","\"A-100\" · \"North\" · 4",[39,1317],{"x":1301,"y":1318,"width":1303,"height":65,"rx":1304,"fill":44,"stroke":138},"112",[56,1320,1322],{"x":1307,"y":1321,"style":1309},"136","row 7",[56,1324,1327],{"x":1307,"y":1325,"style":1326},"160","font-size:12.5px;fill:var(--gold,#b4740a);text-anchor:middle;font-family:ui-monospace,Menlo,monospace","\"a-100 \" · \"north\" · 4.0",[187,1329],{"x1":1330,"y1":1331,"x2":1332,"y2":77,"stroke":1333,"style":1334},"330","67","392","var(--muted,#5b6780)","stroke-width:1.5px",[1336,1337],"polygon",{"points":1338,"fill":1339},"396,100 384,94 386,104","#5b6780",[187,1341],{"x1":1330,"y1":1342,"x2":1332,"y2":1343,"stroke":1333,"style":1334},"145","114",[1336,1345],{"points":1346,"fill":1339},"396,112 384,110 386,120",[39,1348],{"x":1349,"y":1313,"width":1350,"height":1351,"rx":1304,"fill":1352},"398","140","48","#5b5cf0",[56,1354,1358],{"x":1355,"y":1356,"style":1357},"468","103","font-size:12px;font-weight:700;fill:#ffffff;text-anchor:middle","strip · casefold",[56,1360,1362],{"x":1355,"y":1361,"style":1357},"121","round",[187,1364],{"x1":1365,"y1":1366,"x2":1367,"y2":1366,"stroke":1368,"style":54},"542","106","580","var(--teal,#0f9488)",[1336,1370],{"points":1371,"fill":1372},"584,106 572,100 572,112","#0f766e",[39,1374],{"x":1375,"y":1376,"width":58,"height":1377,"rx":1304,"fill":1378,"stroke":1368,"style":54},"586","76","60","#d9f4f1",[56,1380,1384],{"x":1381,"y":1382,"style":1383},"651","100","font-size:11.5px;font-weight:700;fill:var(--teal-ink,#0b6157);text-anchor:middle","same key",[56,1386,1389],{"x":1381,"y":1387,"style":1388},"122","font-size:11px;fill:var(--muted,#5b6780);text-anchor:middle","a-100␟north␟4.0",[56,1391,1394],{"x":107,"y":1392,"style":1393},"204","font-size:12px;fill:var(--muted,#5b6780);text-anchor:middle","Compare on the normalised key; report the original values.",[170,1396,1398],{"id":1397},"group-the-duplicates-into-a-report","Group the duplicates into a report",[10,1400,1401],{},"A flat list of duplicate rows is hard to act on. Grouping them, with counts and the spreadsheet rows involved, is not:",[175,1403,1405],{"className":209,"code":1404,"language":211,"meta":180,"style":180},"import pandas as pd\n\ndef duplicate_report(frame, subset):\n    marked = frame[frame.duplicated(subset=subset, keep=False)].copy()\n    marked[\"Sheet_Row\"] = marked.index + 2\n\n    report = (\n        marked.groupby(subset, dropna=False)\n        .agg(Copies=(\"Sheet_Row\", \"size\"), Rows=(\"Sheet_Row\", lambda s: \", \".join(map(str, s))))\n        .reset_index()\n        .sort_values(\"Copies\", ascending=False)\n    )\n    return report\n\nreport = duplicate_report(df, [\"Order_ID\"])\nprint(report.to_string(index=False))\n\nwith pd.ExcelWriter(\"duplicates.xlsx\", engine=\"openpyxl\") as writer:\n    df.drop(columns=\"_key\").to_excel(writer, sheet_name=\"Orders\", index=False)\n    report.to_excel(writer, sheet_name=\"Duplicates\", index=False)\n",[182,1406,1407,1417,1421,1431,1455,1475,1479,1489,1503,1558,1563,1582,1587,1594,1598,1612,1627,1631,1660,1691],{"__ignoreMap":180},[185,1408,1409,1411,1413,1415],{"class":187,"line":188},[185,1410,219],{"class":218},[185,1412,223],{"class":222},[185,1414,226],{"class":218},[185,1416,229],{"class":222},[185,1418,1419],{"class":187,"line":232},[185,1420,236],{"emptyLinePlaceholder":235},[185,1422,1423,1425,1428],{"class":187,"line":239},[185,1424,743],{"class":218},[185,1426,1427],{"class":746}," duplicate_report",[185,1429,1430],{"class":222},"(frame, subset):\n",[185,1432,1433,1436,1438,1440,1442,1444,1447,1449,1451,1453],{"class":187,"line":245},[185,1434,1435],{"class":222},"    marked ",[185,1437,520],{"class":218},[185,1439,767],{"class":222},[185,1441,770],{"class":516},[185,1443,520],{"class":218},[185,1445,1446],{"class":222},"subset, ",[185,1448,621],{"class":516},[185,1450,520],{"class":218},[185,1452,523],{"class":257},[185,1454,784],{"class":222},[185,1456,1457,1460,1463,1465,1467,1470,1472],{"class":187,"line":295},[185,1458,1459],{"class":222},"    marked[",[185,1461,1462],{"class":195},"\"Sheet_Row\"",[185,1464,915],{"class":222},[185,1466,520],{"class":218},[185,1468,1469],{"class":222}," marked.index ",[185,1471,1129],{"class":218},[185,1473,1474],{"class":257}," 2\n",[185,1476,1477],{"class":187,"line":336},[185,1478,236],{"emptyLinePlaceholder":235},[185,1480,1481,1484,1486],{"class":187,"line":378},[185,1482,1483],{"class":222},"    report ",[185,1485,520],{"class":218},[185,1487,1488],{"class":222}," (\n",[185,1490,1491,1494,1497,1499,1501],{"class":187,"line":419},[185,1492,1493],{"class":222},"        marked.groupby(subset, ",[185,1495,1496],{"class":516},"dropna",[185,1498,520],{"class":218},[185,1500,523],{"class":257},[185,1502,536],{"class":222},[185,1504,1505,1508,1511,1513,1515,1517,1519,1522,1525,1528,1530,1532,1534,1536,1539,1542,1545,1548,1551,1553,1555],{"class":187,"line":461},[185,1506,1507],{"class":222},"        .agg(",[185,1509,1510],{"class":516},"Copies",[185,1512,520],{"class":218},[185,1514,643],{"class":222},[185,1516,1462],{"class":195},[185,1518,261],{"class":222},[185,1520,1521],{"class":195},"\"size\"",[185,1523,1524],{"class":222},"), ",[185,1526,1527],{"class":516},"Rows",[185,1529,520],{"class":218},[185,1531,643],{"class":222},[185,1533,1462],{"class":195},[185,1535,261],{"class":222},[185,1537,1538],{"class":218},"lambda",[185,1540,1541],{"class":222}," s: ",[185,1543,1544],{"class":195},"\", \"",[185,1546,1547],{"class":222},".join(",[185,1549,1550],{"class":257},"map",[185,1552,643],{"class":222},[185,1554,1089],{"class":257},[185,1556,1557],{"class":222},", s))))\n",[185,1559,1560],{"class":187,"line":505},[185,1561,1562],{"class":222},"        .reset_index()\n",[185,1564,1565,1568,1571,1573,1576,1578,1580],{"class":187,"line":853},[185,1566,1567],{"class":222},"        .sort_values(",[185,1569,1570],{"class":195},"\"Copies\"",[185,1572,261],{"class":222},[185,1574,1575],{"class":516},"ascending",[185,1577,520],{"class":218},[185,1579,523],{"class":257},[185,1581,536],{"class":222},[185,1583,1584],{"class":187,"line":859},[185,1585,1586],{"class":222},"    )\n",[185,1588,1589,1591],{"class":187,"line":870},[185,1590,931],{"class":218},[185,1592,1593],{"class":222}," report\n",[185,1595,1596],{"class":187,"line":901},[185,1597,236],{"emptyLinePlaceholder":235},[185,1599,1600,1603,1605,1608,1610],{"class":187,"line":906},[185,1601,1602],{"class":222},"report ",[185,1604,520],{"class":218},[185,1606,1607],{"class":222}," duplicate_report(df, [",[185,1609,251],{"class":195},[185,1611,1212],{"class":222},[185,1613,1614,1616,1619,1621,1623,1625],{"class":187,"line":928},[185,1615,640],{"class":257},[185,1617,1618],{"class":222},"(report.to_string(",[185,1620,517],{"class":516},[185,1622,520],{"class":218},[185,1624,523],{"class":257},[185,1626,1149],{"class":222},[185,1628,1629],{"class":187,"line":937},[185,1630,236],{"emptyLinePlaceholder":235},[185,1632,1633,1636,1639,1642,1644,1647,1649,1652,1655,1657],{"class":187,"line":942},[185,1634,1635],{"class":218},"with",[185,1637,1638],{"class":222}," pd.ExcelWriter(",[185,1640,1641],{"class":195},"\"duplicates.xlsx\"",[185,1643,261],{"class":222},[185,1645,1646],{"class":516},"engine",[185,1648,520],{"class":218},[185,1650,1651],{"class":195},"\"openpyxl\"",[185,1653,1654],{"class":222},") ",[185,1656,226],{"class":218},[185,1658,1659],{"class":222}," writer:\n",[185,1661,1662,1665,1668,1670,1672,1675,1677,1679,1681,1683,1685,1687,1689],{"class":187,"line":953},[185,1663,1664],{"class":222},"    df.drop(",[185,1666,1667],{"class":516},"columns",[185,1669,520],{"class":218},[185,1671,1192],{"class":195},[185,1673,1674],{"class":222},").to_excel(writer, ",[185,1676,528],{"class":516},[185,1678,520],{"class":218},[185,1680,533],{"class":195},[185,1682,261],{"class":222},[185,1684,517],{"class":516},[185,1686,520],{"class":218},[185,1688,523],{"class":257},[185,1690,536],{"class":222},[185,1692,1694,1697,1699,1701,1704,1706,1708,1710,1712],{"class":187,"line":1693},20,[185,1695,1696],{"class":222},"    report.to_excel(writer, ",[185,1698,528],{"class":516},[185,1700,520],{"class":218},[185,1702,1703],{"class":195},"\"Duplicates\"",[185,1705,261],{"class":222},[185,1707,517],{"class":516},[185,1709,520],{"class":218},[185,1711,523],{"class":257},[185,1713,536],{"class":222},[10,1715,1716,1718,1719,1722,1723,1725,1726,1730],{},[182,1717,1527],{}," holding ",[182,1720,1721],{},"\"4, 6\""," tells the reader exactly where to look, and ",[182,1724,1510],{}," sorted descending puts the worst offenders at the top. Pair the sheet with ",[14,1727,1729],{"href":1728},"\u002Fadvanced-data-transformation-and-cleaning\u002Fvalidating-excel-data-with-python\u002Fhighlight-invalid-cells-in-excel-with-python\u002F","highlighting the duplicated cells"," and the fix becomes obvious without any explanation.",[170,1732,1734],{"id":1733},"deciding-which-copy-wins","Deciding which copy wins",[10,1736,1737],{},"When a rule genuinely exists, apply it — and log what was dropped:",[175,1739,1741],{"className":209,"code":1740,"language":211,"meta":180,"style":180},"import pandas as pd\n\nbefore = len(df)\nresolved = (\n    df.sort_values([\"Order_ID\", \"Quantity\"], ascending=[True, False])\n      .drop_duplicates(subset=[\"Order_ID\"], keep=\"first\")     # keep the largest quantity\n)\ndropped = before - len(resolved)\nprint(f\"kept {len(resolved)} row(s), dropped {dropped} duplicate(s)\")\n",[182,1742,1743,1753,1757,1770,1779,1808,1836,1840,1858],{"__ignoreMap":180},[185,1744,1745,1747,1749,1751],{"class":187,"line":188},[185,1746,219],{"class":218},[185,1748,223],{"class":222},[185,1750,226],{"class":218},[185,1752,229],{"class":222},[185,1754,1755],{"class":187,"line":232},[185,1756,236],{"emptyLinePlaceholder":235},[185,1758,1759,1762,1764,1767],{"class":187,"line":239},[185,1760,1761],{"class":222},"before ",[185,1763,520],{"class":218},[185,1765,1766],{"class":257}," len",[185,1768,1769],{"class":222},"(df)\n",[185,1771,1772,1775,1777],{"class":187,"line":245},[185,1773,1774],{"class":222},"resolved ",[185,1776,520],{"class":218},[185,1778,1488],{"class":222},[185,1780,1781,1784,1786,1788,1790,1793,1795,1797,1800,1802,1804,1806],{"class":187,"line":295},[185,1782,1783],{"class":222},"    df.sort_values([",[185,1785,251],{"class":195},[185,1787,261],{"class":222},[185,1789,284],{"class":195},[185,1791,1792],{"class":222},"], ",[185,1794,1575],{"class":516},[185,1796,520],{"class":218},[185,1798,1799],{"class":222},"[",[185,1801,1146],{"class":257},[185,1803,261],{"class":222},[185,1805,523],{"class":257},[185,1807,1212],{"class":222},[185,1809,1810,1813,1815,1817,1819,1821,1823,1825,1827,1830,1833],{"class":187,"line":336},[185,1811,1812],{"class":222},"      .drop_duplicates(",[185,1814,770],{"class":516},[185,1816,520],{"class":218},[185,1818,1799],{"class":222},[185,1820,251],{"class":195},[185,1822,1792],{"class":222},[185,1824,621],{"class":516},[185,1826,520],{"class":218},[185,1828,1829],{"class":195},"\"first\"",[185,1831,1832],{"class":222},")     ",[185,1834,1835],{"class":374},"# keep the largest quantity\n",[185,1837,1838],{"class":187,"line":378},[185,1839,536],{"class":222},[185,1841,1842,1845,1847,1850,1853,1855],{"class":187,"line":419},[185,1843,1844],{"class":222},"dropped ",[185,1846,520],{"class":218},[185,1848,1849],{"class":222}," before ",[185,1851,1852],{"class":218},"-",[185,1854,1766],{"class":257},[185,1856,1857],{"class":222},"(resolved)\n",[185,1859,1860,1862,1864,1866,1869,1871,1873,1876,1878,1881,1883,1886,1888,1891],{"class":187,"line":461},[185,1861,640],{"class":257},[185,1863,643],{"class":222},[185,1865,646],{"class":218},[185,1867,1868],{"class":195},"\"kept ",[185,1870,653],{"class":652},[185,1872,656],{"class":257},[185,1874,1875],{"class":222},"(resolved)",[185,1877,662],{"class":652},[185,1879,1880],{"class":195}," row(s), dropped ",[185,1882,653],{"class":652},[185,1884,1885],{"class":222},"dropped",[185,1887,662],{"class":652},[185,1889,1890],{"class":195}," duplicate(s)\"",[185,1892,536],{"class":222},[10,1894,1895,1896,1899,1900,1904],{},"Sorting before ",[182,1897,1898],{},"drop_duplicates"," is what turns \"keep the first\" into a real rule — here, keep the largest quantity per order. Without the sort, \"first\" means \"whichever row happened to be higher in the file\", which is not a rule at all. Whenever the choice is arbitrary, resist the temptation: report the group and let the data owner decide, because a silently discarded row is impossible to notice later. The ",[14,1901,1903],{"href":1902},"\u002Fadvanced-data-transformation-and-cleaning\u002Fcleaning-excel-data-with-pandas\u002Fpandas-drop-duplicates-from-excel-column\u002F","drop duplicates from an Excel column"," guide goes through the resolution strategies in more detail.",[170,1906,1908],{"id":1907},"where-duplicates-come-from","Where duplicates come from",[10,1910,1911],{},"Knowing the source usually decides the fix, and there are only a few common origins:",[20,1913,29,1918,29,1921,29,1924,29,1927,29,1932,29,1937,29,1942,29,1947,29,1950,29,1954,29,1957,29,1960,29,1962,29,1967,29,1970,29,1974,29,1976,29,1980,29,1983],{"viewBox":1914,"role":23,"ariaLabelledBy":1915,"xmlns":27,"style":28},"0 0 740 214",[1916,1917],"dup-src-t","dup-src-d",[31,1919,1920],{"id":1916},"Four common origins of duplicate rows and the fix for each",[35,1922,1923],{"id":1917},"A re-run import duplicates every row and is fixed by a load key. A copy-paste duplicates a block and is fixed by dropping exact copies. Two source systems produce conflicting keys and need a reconciliation rule. Manual re-entry produces near duplicates and is fixed by normalising before comparing.",[39,1925],{"x":41,"y":41,"width":42,"height":1926,"fill":44},"214",[39,1928],{"x":1929,"y":48,"width":1930,"height":1376,"rx":1931,"fill":52,"stroke":69},"20","340","12",[56,1933,1936],{"x":1934,"y":59,"style":1935},"40","font-size:12.5px;font-weight:700;fill:var(--brand-strong,#4338ca)","import run twice",[56,1938,1941],{"x":1934,"y":1939,"style":1940},"74","font-size:11.5px;fill:var(--muted,#5b6780)","every row appears exactly twice",[56,1943,1946],{"x":1934,"y":1944,"style":1945},"92","font-size:11.5px;fill:var(--teal,#0f9488)","fix: a load key, or drop exact copies",[39,1948],{"x":1949,"y":48,"width":1930,"height":1376,"rx":1931,"fill":52,"stroke":69},"380",[56,1951,1953],{"x":1952,"y":59,"style":1935},"400","copy-paste in the sheet",[56,1955,1956],{"x":1952,"y":1939,"style":1940},"a contiguous block repeats",[56,1958,1959],{"x":1952,"y":1944,"style":1945},"fix: drop exact copies, keep first",[39,1961],{"x":1929,"y":1318,"width":1930,"height":1376,"rx":1931,"fill":103,"stroke":104},[56,1963,1966],{"x":1934,"y":1964,"style":1965},"138","font-size:12.5px;font-weight:700;fill:var(--danger,#dc2626)","two systems, one key",[56,1968,1969],{"x":1934,"y":1325,"style":1940},"same id, different values",[56,1971,1973],{"x":1934,"y":89,"style":1972},"font-size:11.5px;fill:var(--danger,#dc2626)","fix: a reconciliation rule, or ask",[39,1975],{"x":1949,"y":1318,"width":1930,"height":1376,"rx":1931,"fill":137,"stroke":138},[56,1977,1979],{"x":1952,"y":1964,"style":1978},"font-size:12.5px;font-weight:700;fill:var(--gold,#b4740a)","re-entered by hand",[56,1981,1982],{"x":1952,"y":1325,"style":1940},"differs by case, spacing, spelling",[56,1984,1985],{"x":1952,"y":89,"style":1945},"fix: normalise, then re-check",[10,1987,1988],{},"The bottom-left box is the only one that cannot be solved in code alone. When two systems disagree about the same order, choosing a winner is a business decision — and encoding that decision as \"keep the row from system A\" is fine, provided it is written down somewhere other than a sort key buried in a script.",[170,1990,1992],{"id":1991},"common-pitfalls-and-gotchas","Common pitfalls and gotchas",[1994,1995,1996,2012],"table",{},[1997,1998,1999],"thead",{},[2000,2001,2002,2006,2009],"tr",{},[2003,2004,2005],"th",{},"Symptom",[2003,2007,2008],{},"Cause",[2003,2010,2011],{},"Fix",[2013,2014,2015,2027,2044,2055,2066,2079],"tbody",{},[2000,2016,2017,2021,2024],{},[2018,2019,2020],"td",{},"Obvious duplicates not detected",[2018,2022,2023],{},"Whitespace, case or float differences",[2018,2025,2026],{},"Compare on a normalised key",[2000,2028,2029,2032,2038],{},[2018,2030,2031],{},"Only half of each pair reported",[2018,2033,2034,2035],{},"Default ",[182,2036,2037],{},"keep=\"first\"",[2018,2039,2040,2041,2043],{},"Use ",[182,2042,549],{}," for reporting",[2000,2045,2046,2049,2052],{},[2018,2047,2048],{},"Every row flagged as duplicate",[2018,2050,2051],{},"Compared on too few columns",[2018,2053,2054],{},"Include the columns that make a row unique",[2000,2056,2057,2060,2063],{},[2018,2058,2059],{},"Duplicates reappear next run",[2018,2061,2062],{},"Source system re-sends the same rows",[2018,2064,2065],{},"Deduplicate on a stable business key, not on load order",[2000,2067,2068,2071,2076],{},[2018,2069,2070],{},"Wrong copy kept",[2018,2072,2073,2075],{},[182,2074,1898],{}," without a deliberate sort",[2018,2077,2078],{},"Sort by the tie-breaking column first",[2000,2080,2081,2084,2093],{},[2018,2082,2083],{},"Blank keys grouped together",[2018,2085,2086,2089,2090],{},[182,2087,2088],{},"NaN"," treated as a value with ",[182,2091,2092],{},"dropna=False",[2018,2094,2095],{},"Validate that the key is present before grouping",[170,2097,2099],{"id":2098},"check-for-duplicates-before-you-join","Check for duplicates before you join",[10,2101,2102],{},"The most expensive duplicate is the one nobody looked for before a merge. Joining a 500-row orders table to a customer table whose key repeats three times produces 1,500 rows and a revenue figure three times too large — and the merge itself raises nothing. Assert uniqueness on the side that is supposed to be unique, and let pandas do the checking:",[175,2104,2106],{"className":209,"code":2105,"language":211,"meta":180,"style":180},"import pandas as pd\n\norders = pd.read_excel(\"orders.xlsx\", sheet_name=\"Orders\")\ncustomers = pd.read_excel(\"customers.xlsx\")\n\n# validate=\"m:1\" raises if the right-hand key is not unique\nenriched = orders.merge(customers, on=\"Customer_ID\", how=\"left\", validate=\"m:1\")\nprint(len(orders), \"->\", len(enriched))\n",[182,2107,2108,2118,2122,2143,2157,2161,2166,2206],{"__ignoreMap":180},[185,2109,2110,2112,2114,2116],{"class":187,"line":188},[185,2111,219],{"class":218},[185,2113,223],{"class":222},[185,2115,226],{"class":218},[185,2117,229],{"class":222},[185,2119,2120],{"class":187,"line":232},[185,2121,236],{"emptyLinePlaceholder":235},[185,2123,2124,2127,2129,2131,2133,2135,2137,2139,2141],{"class":187,"line":239},[185,2125,2126],{"class":222},"orders ",[185,2128,520],{"class":218},[185,2130,579],{"class":222},[185,2132,511],{"class":195},[185,2134,261],{"class":222},[185,2136,528],{"class":516},[185,2138,520],{"class":218},[185,2140,533],{"class":195},[185,2142,536],{"class":222},[185,2144,2145,2148,2150,2152,2155],{"class":187,"line":245},[185,2146,2147],{"class":222},"customers ",[185,2149,520],{"class":218},[185,2151,579],{"class":222},[185,2153,2154],{"class":195},"\"customers.xlsx\"",[185,2156,536],{"class":222},[185,2158,2159],{"class":187,"line":295},[185,2160,236],{"emptyLinePlaceholder":235},[185,2162,2163],{"class":187,"line":336},[185,2164,2165],{"class":374},"# validate=\"m:1\" raises if the right-hand key is not unique\n",[185,2167,2168,2171,2173,2176,2179,2181,2184,2186,2189,2191,2194,2196,2199,2201,2204],{"class":187,"line":378},[185,2169,2170],{"class":222},"enriched ",[185,2172,520],{"class":218},[185,2174,2175],{"class":222}," orders.merge(customers, ",[185,2177,2178],{"class":516},"on",[185,2180,520],{"class":218},[185,2182,2183],{"class":195},"\"Customer_ID\"",[185,2185,261],{"class":222},[185,2187,2188],{"class":516},"how",[185,2190,520],{"class":218},[185,2192,2193],{"class":195},"\"left\"",[185,2195,261],{"class":222},[185,2197,2198],{"class":516},"validate",[185,2200,520],{"class":218},[185,2202,2203],{"class":195},"\"m:1\"",[185,2205,536],{"class":222},[185,2207,2208,2210,2212,2214,2217,2220,2222,2224],{"class":187,"line":419},[185,2209,640],{"class":257},[185,2211,643],{"class":222},[185,2213,656],{"class":257},[185,2215,2216],{"class":222},"(orders), ",[185,2218,2219],{"class":195},"\"->\"",[185,2221,261],{"class":222},[185,2223,656],{"class":257},[185,2225,2226],{"class":222},"(enriched))\n",[10,2228,2229,2232,2233,2236,2237,1279,2240,2243],{},[182,2230,2231],{},"validate=\"m:1\""," turns a silent row explosion into an immediate ",[182,2234,2235],{},"MergeError",", which is exactly the trade you want: a job that fails on Monday morning costs an hour, and a report that overstates revenue by a factor of three costs considerably more. The same argument accepts ",[182,2238,2239],{},"\"1:1\"",[182,2241,2242],{},"\"1:m\""," for the other shapes, and it is worth adding to every merge in a reporting pipeline rather than only to the ones that have already burned you.",[170,2245,2247],{"id":2246},"performance-and-scale-notes","Performance and scale notes",[10,2249,2250,2252,2253,2256],{},[182,2251,545],{}," hashes rows, so it is roughly linear and fast even on hundreds of thousands of rows. Building a normalised key with string operations is the expensive part — several passes over the column — so only normalise the columns that form the key rather than the whole frame. The ",[182,2254,2255],{},"groupby"," report is cheap by comparison. On very large sheets, deduplicate on a narrow key first and only then compare the full rows within the small set of groups that survived, which avoids building a long concatenated key for every row in the file.",[170,2258,2260],{"id":2259},"conclusion","Conclusion",[10,2262,2263,2264,2267,2268,2270],{},"\"Duplicate\" means three different things, and only one of them is safe to fix automatically. Use ",[182,2265,2266],{},"duplicated(keep=False)"," to report, compare on a normalised key so near-duplicates cannot hide, separate repeated keys with conflicting data from honest copies with ",[182,2269,981],{},", and group the results with counts and sheet-row numbers. Delete only where a deliberate sort expresses a real rule about which copy wins.",[170,2272,2274],{"id":2273},"frequently-asked-questions","Frequently asked questions",[10,2276,2277,2281],{},[2278,2279,2280],"strong",{},"What is the difference between a duplicate row and a duplicate key?","\nA duplicate row repeats every column, and is usually a copy-paste or a double import. A duplicate key repeats an identifier while other columns differ, which means two records disagree about the same thing — a data problem, not a copy.",[10,2283,2284,2290],{},[2278,2285,2286,2287,2289],{},"Why does ",[182,2288,545],{}," miss obvious duplicates?","\nIt compares values exactly. Trailing spaces, different capitalisation, a number stored as text and a float rounding difference all make two visually identical rows unequal.",[10,2292,2293,2296,2297,2299,2300,2302],{},[2278,2294,2295],{},"How do I see every copy rather than just the repeats?","\nPass ",[182,2298,549],{}," to ",[182,2301,545],{},", which marks all members of each duplicate group including the first.",[10,2304,2305,2308],{},[2278,2306,2307],{},"Should I delete duplicates automatically?","\nOnly when the rule for which copy wins is unambiguous. Otherwise report the groups and let the data owner decide, because deleting the wrong copy is invisible afterwards.",[170,2310,2312],{"id":2311},"related","Related",[10,2314,2315],{},"Up to the parent guide:",[2317,2318,2319],"ul",{},[2320,2321,2322,2324],"li",{},[14,2323,17],{"href":16}," — the cross-row checks Excel cannot perform.",[10,2326,2327],{},"Related guides:",[2317,2329,2330,2336,2343,2349],{},[2320,2331,2332,2335],{},[14,2333,2334],{"href":1902},"Drop Duplicates from an Excel Column with pandas"," — the resolution side, in depth.",[2320,2337,2338,2342],{},[14,2339,2341],{"href":2340},"\u002Fadvanced-data-transformation-and-cleaning\u002Fvalidating-excel-data-with-python\u002Fcompare-two-excel-files-for-differences-with-python\u002F","Compare Two Excel Files for Differences with Python"," — duplicates across files rather than within one.",[2320,2344,2345,2348],{},[14,2346,2347],{"href":1728},"Highlight Invalid Cells in Excel with Python"," — showing the duplicate rows in the workbook.",[2320,2350,2351,2355],{},[14,2352,2354],{"href":2353},"\u002Fadvanced-data-transformation-and-cleaning\u002Fmerging-and-joining-excel-dataframes\u002F","Merging and Joining Excel DataFrames"," — where an undetected duplicate key multiplies rows.",[2357,2358,2359],"style",{},"html pre.shiki code .sMTad, html code.shiki .sMTad{--shiki-default:#6F42C1;--shiki-dark:#FFB757}html pre.shiki code .srMev, html code.shiki .srMev{--shiki-default:#032F62;--shiki-dark:#ADDCFF}html .default .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .dark .shiki span {color: var(--shiki-dark);background: var(--shiki-dark-bg);font-style: var(--shiki-dark-font-style);font-weight: var(--shiki-dark-font-weight);text-decoration: var(--shiki-dark-text-decoration);}html.dark .shiki span {color: var(--shiki-dark);background: var(--shiki-dark-bg);font-style: var(--shiki-dark-font-style);font-weight: var(--shiki-dark-font-weight);text-decoration: var(--shiki-dark-text-decoration);}html pre.shiki code .s-kum, html code.shiki .s-kum{--shiki-default:#D73A49;--shiki-dark:#FF9492}html pre.shiki code .skGVy, html code.shiki .skGVy{--shiki-default:#24292E;--shiki-dark:#F0F3F6}html pre.shiki code .sP0c6, html code.shiki .sP0c6{--shiki-default:#005CC5;--shiki-dark:#91CBFF}html pre.shiki code .s-wDw, html code.shiki .s-wDw{--shiki-default:#6A737D;--shiki-dark:#BDC4CC}html pre.shiki code .sa561, html code.shiki .sa561{--shiki-default:#E36209;--shiki-dark:#FFB757}html pre.shiki code .sSjpA, html code.shiki .sSjpA{--shiki-default:#005CC5;--shiki-dark:#FF9492}html pre.shiki code .s_Opv, html code.shiki .s_Opv{--shiki-default:#6F42C1;--shiki-dark:#DBB7FF}",{"title":180,"searchDepth":232,"depth":232,"links":2361},[2362,2363,2364,2365,2366,2367,2368,2369,2370,2371,2372,2373,2374,2375],{"id":172,"depth":232,"text":173},{"id":205,"depth":232,"text":206},{"id":539,"depth":232,"text":540},{"id":695,"depth":232,"text":696},{"id":985,"depth":232,"text":986},{"id":1397,"depth":232,"text":1398},{"id":1733,"depth":232,"text":1734},{"id":1907,"depth":232,"text":1908},{"id":1991,"depth":232,"text":1992},{"id":2098,"depth":232,"text":2099},{"id":2246,"depth":232,"text":2247},{"id":2259,"depth":232,"text":2260},{"id":2273,"depth":232,"text":2274},{"id":2311,"depth":232,"text":2312},"2026-08-01","Detect exact and near-duplicate rows in a spreadsheet with pandas — key-based duplicates, whitespace and case traps, duplicate groups with counts, and a report you can send back.","md",[2380,2382,2385,2387],{"q":2280,"a":2381},"A duplicate row repeats every column, and is usually a copy-paste or a double import. A duplicate key repeats an identifier while other columns differ, which means two records disagree about the same thing — a data problem, not a copy.",{"q":2383,"a":2384},"Why does duplicated() miss obvious duplicates?","It compares values exactly. Trailing spaces, different capitalisation, a number stored as text and a float rounding difference all make two visually identical rows unequal.",{"q":2295,"a":2386},"Pass keep=False to duplicated(), which marks all members of each duplicate group including the first.",{"q":2307,"a":2388},"Only when the rule for which copy wins is unambiguous. Otherwise report the groups and let the data owner decide, because deleting the wrong copy is invisible afterwards.",{"breadcrumb":2390},[2391,2394,2397,2398],{"name":2392,"item":2393},"Home","\u002F",{"name":2395,"item":2396},"Advanced Data Transformation and Cleaning","\u002Fadvanced-data-transformation-and-cleaning\u002F",{"name":17,"item":16},{"name":5,"item":2399},"\u002Fadvanced-data-transformation-and-cleaning\u002Fvalidating-excel-data-with-python\u002Ffind-duplicate-rows-in-excel-with-python\u002F","\u002Fadvanced-data-transformation-and-cleaning\u002Fvalidating-excel-data-with-python\u002Ffind-duplicate-rows-in-excel-with-python",{"title":5,"description":2402},"Use duplicated() and groupby to find repeated rows and repeated keys in Excel data, normalise before comparing, and produce a duplicates report with row numbers and counts.","find-duplicate-rows-in-excel-with-python","advanced-data-transformation-and-cleaning\u002Fvalidating-excel-data-with-python\u002Ffind-duplicate-rows-in-excel-with-python\u002Findex","how-to","-9GhBesnXL11vZ8uZx0uE4sTVKJUK7CNwRFh9BP2Dz8",[2408,2411],{"title":2341,"path":2409,"stem":2410,"children":-1},"\u002Fadvanced-data-transformation-and-cleaning\u002Fvalidating-excel-data-with-python\u002Fcompare-two-excel-files-for-differences-with-python","advanced-data-transformation-and-cleaning\u002Fvalidating-excel-data-with-python\u002Fcompare-two-excel-files-for-differences-with-python\u002Findex",{"title":2347,"path":2412,"stem":2413,"children":-1},"\u002Fadvanced-data-transformation-and-cleaning\u002Fvalidating-excel-data-with-python\u002Fhighlight-invalid-cells-in-excel-with-python","advanced-data-transformation-and-cleaning\u002Fvalidating-excel-data-with-python\u002Fhighlight-invalid-cells-in-excel-with-python\u002Findex",1785584462739]