[{"data":1,"prerenderedAt":2341},["ShallowReactive",2],{"doc:\u002Fadvanced-data-transformation-and-cleaning\u002Fvalidating-excel-data-with-python\u002Fvalidate-excel-columns-before-import-with-pandas":3,"surround:\u002Fadvanced-data-transformation-and-cleaning\u002Fvalidating-excel-data-with-python\u002Fvalidate-excel-columns-before-import-with-pandas":2332},{"id":4,"title":5,"body":6,"dateModified":2302,"datePublished":2302,"description":2303,"extension":2304,"faq":2305,"meta":2314,"navigation":263,"path":2325,"seo":2326,"slug":2328,"stem":2329,"type":2330,"__hash__":2331},"docs\u002Fadvanced-data-transformation-and-cleaning\u002Fvalidating-excel-data-with-python\u002Fvalidate-excel-columns-before-import-with-pandas\u002Findex.md","Validate Excel Columns Before Import with pandas",{"type":7,"value":8,"toc":2287},"minimark",[9,28,193,198,225,229,235,516,528,532,535,788,795,799,808,1022,1040,1155,1159,1162,1512,1519,1523,1530,1800,1811,1815,1972,1975,1979,1982,2064,2067,2071,2174,2178,2192,2196,2199,2203,2210,2216,2222,2231,2235,2238,2247,2250,2283],[10,11,12,13,17,18,21,22,27],"p",{},"Most import failures are not subtle. The sheet was renamed, someone inserted a column, or a header reads ",[14,15,16],"code",{},"Qty"," this month and ",[14,19,20],{},"Quantity"," last month. Catching those before any row is processed turns a confusing downstream error into a one-line message a submitter can act on. This guide, part of ",[23,24,26],"a",{"href":25},"\u002Fadvanced-data-transformation-and-cleaning\u002Fvalidating-excel-data-with-python\u002F","Validating Excel Data with Python",", builds that structural gate.",[29,30,38,39,38,43,38,47,38,54,38,63,38,70,38,74,38,82,38,87,38,96,38,102,38,106,38,110,38,113,38,117,38,120,38,123,38,126,38,130,38,133,38,136,38,139,38,143,38,146,38,149,38,154,38,158,38,163,38,170,38,176,38,181,38,183,38,185,38,187],"svg",{"viewBox":31,"role":32,"ariaLabelledBy":33,"xmlns":36,"style":37},"0 0 740 236","img",[34,35],"vc-gate-t","vc-gate-d","http:\u002F\u002Fwww.w3.org\u002F2000\u002Fsvg","width:100%;max-width:740px;height:auto;display:block;margin:1.5rem auto;font-family:Inter,ui-sans-serif,system-ui,sans-serif","\n  ",[40,41,42],"title",{"id":34},"A structural gate in front of the import",[44,45,46],"desc",{"id":35},"Four checks run in order before any row is read: does the sheet exist, do the headers normalise to known names, are all required columns present, and is there at least one data row. Failing any check stops the import with a specific message.",[48,49],"rect",{"x":50,"y":50,"width":51,"height":52,"fill":53},"0","740","236","#ffffff",[48,55],{"x":56,"y":57,"width":58,"height":59,"rx":60,"fill":61,"stroke":62},"20","86","120","60","12","#ebebfd","var(--line,#cdd5e6)",[64,65,69],"text",{"x":66,"y":67,"style":68},"80","112","font-size:12px;font-weight:700;fill:var(--brand-strong,#4338ca);text-anchor:middle","incoming",[64,71,73],{"x":66,"y":72,"style":68},"132","workbook",[75,76],"line",{"x1":77,"y1":78,"x2":79,"y2":78,"stroke":80,"style":81},"144","116","168","var(--muted,#5b6780)","stroke-width:2px",[83,84],"polygon",{"points":85,"fill":86},"172,116 160,110 160,122","#5b6780",[48,88],{"x":89,"y":90,"width":91,"height":92,"rx":93,"fill":94,"stroke":95},"174","70","118","92","10","#f0f4ff","var(--brand,#5b5cf0)",[64,97,101],{"x":98,"y":99,"style":100},"233","100","font-size:12px;font-weight:700;fill:var(--text,#172033);text-anchor:middle","sheet",[64,103,105],{"x":98,"y":58,"style":104},"font-size:11px;fill:var(--muted,#5b6780);text-anchor:middle","named as",[64,107,109],{"x":98,"y":108,"style":104},"138","expected?",[48,111],{"x":112,"y":90,"width":91,"height":92,"rx":93,"fill":94,"stroke":95},"300",[64,114,116],{"x":115,"y":99,"style":100},"359","headers",[64,118,119],{"x":115,"y":58,"style":104},"normalise +",[64,121,122],{"x":115,"y":108,"style":104},"map aliases",[48,124],{"x":125,"y":90,"width":91,"height":92,"rx":93,"fill":94,"stroke":95},"426",[64,127,129],{"x":128,"y":99,"style":100},"485","required",[64,131,132],{"x":128,"y":58,"style":104},"columns all",[64,134,135],{"x":128,"y":108,"style":104},"present?",[48,137],{"x":138,"y":90,"width":91,"height":92,"rx":93,"fill":94,"stroke":95},"552",[64,140,142],{"x":141,"y":99,"style":100},"611","rows",[64,144,145],{"x":141,"y":58,"style":104},"at least one",[64,147,148],{"x":141,"y":108,"style":104},"data row?",[75,150],{"x1":151,"y1":78,"x2":152,"y2":78,"stroke":153,"style":81},"674","704","var(--teal,#0f9488)",[83,155],{"points":156,"fill":157},"708,116 696,110 696,122","#0f766e",[64,159,162],{"x":160,"y":58,"style":161},"712","font-size:12px;font-weight:700;fill:var(--teal,#0f9488);text-anchor:end"," ",[48,164],{"x":89,"y":165,"width":166,"height":167,"rx":93,"fill":168,"stroke":169},"182","496","40","#fce9e9","var(--danger,#dc2626)",[64,171,175],{"x":172,"y":173,"style":174},"422","207","font-size:12px;font-weight:700;fill:var(--danger,#dc2626);text-anchor:middle","any failure → stop, name the file, the sheet and the exact problem",[75,177],{"x1":98,"y1":178,"x2":98,"y2":179,"stroke":169,"style":180},"166","178","stroke-width:1.5px",[75,182],{"x1":115,"y1":178,"x2":115,"y2":179,"stroke":169,"style":180},[75,184],{"x1":128,"y1":178,"x2":128,"y2":179,"stroke":169,"style":180},[75,186],{"x1":141,"y1":178,"x2":141,"y2":179,"stroke":169,"style":180},[64,188,192],{"x":189,"y":190,"style":191},"370","42","font-size:12.5px;font-weight:600;fill:var(--muted,#5b6780);text-anchor:middle","Structure first — no row is worth reading until the shape is right.",[194,195,197],"h2",{"id":196},"prerequisites","Prerequisites",[199,200,205],"pre",{"className":201,"code":202,"language":203,"meta":204,"style":204},"language-bash shiki shiki-themes github-light github-dark-high-contrast","pip install pandas openpyxl\n","bash","",[14,206,207],{"__ignoreMap":204},[208,209,211,215,219,222],"span",{"class":75,"line":210},1,[208,212,214],{"class":213},"sMTad","pip",[208,216,218],{"class":217},"srMev"," install",[208,220,221],{"class":217}," pandas",[208,223,224],{"class":217}," openpyxl\n",[194,226,228],{"id":227},"check-the-sheet-before-reading-it","Check the sheet before reading it",[10,230,231,234],{},[14,232,233],{},"pd.ExcelFile"," reads only the workbook's structure, so you can inspect sheet names without loading a single row — useful when the file is large and the sheet may not be there at all:",[199,236,240],{"className":237,"code":238,"language":239,"meta":204,"style":204},"language-python shiki shiki-themes github-light github-dark-high-contrast","import pandas as pd\n\ndef open_sheet(path, sheet=\"Orders\"):\n    book = pd.ExcelFile(path)\n    if sheet not in book.sheet_names:\n        close = [s for s in book.sheet_names if s.strip().lower() == sheet.lower()]\n        if close:\n            raise ValueError(\n                f\"sheet {sheet!r} not found, but {close[0]!r} looks like it \"\n                \"— fix the tab name or update the importer\"\n            )\n        raise ValueError(f\"sheet {sheet!r} not found; file contains {book.sheet_names}\")\n    return book.parse(sheet, dtype=object)\n\ndf = open_sheet(\"submitted.xlsx\")\nprint(df.shape)\n","python",[14,241,242,258,265,287,298,316,351,360,373,414,420,426,466,486,491,507],{"__ignoreMap":204},[208,243,244,248,252,255],{"class":75,"line":210},[208,245,247],{"class":246},"s-kum","import",[208,249,251],{"class":250},"skGVy"," pandas ",[208,253,254],{"class":246},"as",[208,256,257],{"class":250}," pd\n",[208,259,261],{"class":75,"line":260},2,[208,262,264],{"emptyLinePlaceholder":263},true,"\n",[208,266,268,271,275,278,281,284],{"class":75,"line":267},3,[208,269,270],{"class":246},"def",[208,272,274],{"class":273},"s_Opv"," open_sheet",[208,276,277],{"class":250},"(path, sheet",[208,279,280],{"class":246},"=",[208,282,283],{"class":217},"\"Orders\"",[208,285,286],{"class":250},"):\n",[208,288,290,293,295],{"class":75,"line":289},4,[208,291,292],{"class":250},"    book ",[208,294,280],{"class":246},[208,296,297],{"class":250}," pd.ExcelFile(path)\n",[208,299,301,304,307,310,313],{"class":75,"line":300},5,[208,302,303],{"class":246},"    if",[208,305,306],{"class":250}," sheet ",[208,308,309],{"class":246},"not",[208,311,312],{"class":246}," in",[208,314,315],{"class":250}," book.sheet_names:\n",[208,317,319,322,324,327,330,333,336,339,342,345,348],{"class":75,"line":318},6,[208,320,321],{"class":250},"        close ",[208,323,280],{"class":246},[208,325,326],{"class":250}," [s ",[208,328,329],{"class":246},"for",[208,331,332],{"class":250}," s ",[208,334,335],{"class":246},"in",[208,337,338],{"class":250}," book.sheet_names ",[208,340,341],{"class":246},"if",[208,343,344],{"class":250}," s.strip().lower() ",[208,346,347],{"class":246},"==",[208,349,350],{"class":250}," sheet.lower()]\n",[208,352,354,357],{"class":75,"line":353},7,[208,355,356],{"class":246},"        if",[208,358,359],{"class":250}," close:\n",[208,361,363,366,370],{"class":75,"line":362},8,[208,364,365],{"class":246},"            raise",[208,367,369],{"class":368},"sP0c6"," ValueError",[208,371,372],{"class":250},"(\n",[208,374,376,379,382,386,388,391,394,397,399,402,404,407,409,411],{"class":75,"line":375},9,[208,377,378],{"class":246},"                f",[208,380,381],{"class":217},"\"sheet ",[208,383,385],{"class":384},"sSjpA","{",[208,387,101],{"class":250},[208,389,390],{"class":246},"!r",[208,392,393],{"class":384},"}",[208,395,396],{"class":217}," not found, but ",[208,398,385],{"class":384},[208,400,401],{"class":250},"close[",[208,403,50],{"class":368},[208,405,406],{"class":250},"]",[208,408,390],{"class":246},[208,410,393],{"class":384},[208,412,413],{"class":217}," looks like it \"\n",[208,415,417],{"class":75,"line":416},10,[208,418,419],{"class":217},"                \"— fix the tab name or update the importer\"\n",[208,421,423],{"class":75,"line":422},11,[208,424,425],{"class":250},"            )\n",[208,427,429,432,434,437,440,442,444,446,448,450,453,455,458,460,463],{"class":75,"line":428},12,[208,430,431],{"class":246},"        raise",[208,433,369],{"class":368},[208,435,436],{"class":250},"(",[208,438,439],{"class":246},"f",[208,441,381],{"class":217},[208,443,385],{"class":384},[208,445,101],{"class":250},[208,447,390],{"class":246},[208,449,393],{"class":384},[208,451,452],{"class":217}," not found; file contains ",[208,454,385],{"class":384},[208,456,457],{"class":250},"book.sheet_names",[208,459,393],{"class":384},[208,461,462],{"class":217},"\"",[208,464,465],{"class":250},")\n",[208,467,469,472,475,479,481,484],{"class":75,"line":468},13,[208,470,471],{"class":246},"    return",[208,473,474],{"class":250}," book.parse(sheet, ",[208,476,478],{"class":477},"sa561","dtype",[208,480,280],{"class":246},[208,482,483],{"class":368},"object",[208,485,465],{"class":250},[208,487,489],{"class":75,"line":488},14,[208,490,264],{"emptyLinePlaceholder":263},[208,492,494,497,499,502,505],{"class":75,"line":493},15,[208,495,496],{"class":250},"df ",[208,498,280],{"class":246},[208,500,501],{"class":250}," open_sheet(",[208,503,504],{"class":217},"\"submitted.xlsx\"",[208,506,465],{"class":250},[208,508,510,513],{"class":75,"line":509},16,[208,511,512],{"class":368},"print",[208,514,515],{"class":250},"(df.shape)\n",[10,517,518,519,523,524,527],{},"Listing the sheets that ",[520,521,522],"em",{},"are"," present turns \"file is wrong\" into \"you sent the Q1 tab, we need Orders\". The near-match branch handles the most common version of the problem — a tab called ",[14,525,526],{},"orders "," with a trailing space — without silently accepting it.",[194,529,531],{"id":530},"normalise-headers-before-comparing-them","Normalise headers before comparing them",[10,533,534],{},"Header text arrives with stray spaces, non-breaking spaces, line breaks from wrapped cells and inconsistent case. Normalise first, then compare:",[199,536,538],{"className":237,"code":537,"language":239,"meta":204,"style":204},"import re\nimport pandas as pd\n\ndef normalise(name):\n    text = str(name).replace(\"\\xa0\", \" \")          # non-breaking space\n    text = re.sub(r\"\\s+\", \" \", text).strip()       # collapse whitespace incl. newlines\n    return text.lower().replace(\" \", \"_\")\n\ndef normalise_headers(df):\n    df = df.copy()\n    df.columns = [normalise(c) for c in df.columns]\n    unnamed = [c for c in df.columns if c.startswith(\"unnamed:\")]\n    if unnamed:\n        df = df.drop(columns=unnamed)              # blank columns pandas auto-named\n    return df\n\ndf = normalise_headers(df)\nprint(list(df.columns))\n",[14,539,540,547,557,561,571,604,636,652,656,666,676,696,726,733,754,761,765,775],{"__ignoreMap":204},[208,541,542,544],{"class":75,"line":210},[208,543,247],{"class":246},[208,545,546],{"class":250}," re\n",[208,548,549,551,553,555],{"class":75,"line":260},[208,550,247],{"class":246},[208,552,251],{"class":250},[208,554,254],{"class":246},[208,556,257],{"class":250},[208,558,559],{"class":75,"line":267},[208,560,264],{"emptyLinePlaceholder":263},[208,562,563,565,568],{"class":75,"line":289},[208,564,270],{"class":246},[208,566,567],{"class":273}," normalise",[208,569,570],{"class":250},"(name):\n",[208,572,573,576,578,581,584,586,589,591,594,597,600],{"class":75,"line":300},[208,574,575],{"class":250},"    text ",[208,577,280],{"class":246},[208,579,580],{"class":368}," str",[208,582,583],{"class":250},"(name).replace(",[208,585,462],{"class":217},[208,587,588],{"class":384},"\\xa0",[208,590,462],{"class":217},[208,592,593],{"class":250},", ",[208,595,596],{"class":217},"\" \"",[208,598,599],{"class":250},")          ",[208,601,603],{"class":602},"s-wDw","# non-breaking space\n",[208,605,606,608,610,613,616,618,621,624,626,628,630,633],{"class":75,"line":318},[208,607,575],{"class":250},[208,609,280],{"class":246},[208,611,612],{"class":250}," re.sub(",[208,614,615],{"class":246},"r",[208,617,462],{"class":217},[208,619,620],{"class":368},"\\s",[208,622,623],{"class":246},"+",[208,625,462],{"class":217},[208,627,593],{"class":250},[208,629,596],{"class":217},[208,631,632],{"class":250},", text).strip()       ",[208,634,635],{"class":602},"# collapse whitespace incl. newlines\n",[208,637,638,640,643,645,647,650],{"class":75,"line":353},[208,639,471],{"class":246},[208,641,642],{"class":250}," text.lower().replace(",[208,644,596],{"class":217},[208,646,593],{"class":250},[208,648,649],{"class":217},"\"_\"",[208,651,465],{"class":250},[208,653,654],{"class":75,"line":362},[208,655,264],{"emptyLinePlaceholder":263},[208,657,658,660,663],{"class":75,"line":375},[208,659,270],{"class":246},[208,661,662],{"class":273}," normalise_headers",[208,664,665],{"class":250},"(df):\n",[208,667,668,671,673],{"class":75,"line":416},[208,669,670],{"class":250},"    df ",[208,672,280],{"class":246},[208,674,675],{"class":250}," df.copy()\n",[208,677,678,681,683,686,688,691,693],{"class":75,"line":422},[208,679,680],{"class":250},"    df.columns ",[208,682,280],{"class":246},[208,684,685],{"class":250}," [normalise(c) ",[208,687,329],{"class":246},[208,689,690],{"class":250}," c ",[208,692,335],{"class":246},[208,694,695],{"class":250}," df.columns]\n",[208,697,698,701,703,706,708,710,712,715,717,720,723],{"class":75,"line":428},[208,699,700],{"class":250},"    unnamed ",[208,702,280],{"class":246},[208,704,705],{"class":250}," [c ",[208,707,329],{"class":246},[208,709,690],{"class":250},[208,711,335],{"class":246},[208,713,714],{"class":250}," df.columns ",[208,716,341],{"class":246},[208,718,719],{"class":250}," c.startswith(",[208,721,722],{"class":217},"\"unnamed:\"",[208,724,725],{"class":250},")]\n",[208,727,728,730],{"class":75,"line":468},[208,729,303],{"class":246},[208,731,732],{"class":250}," unnamed:\n",[208,734,735,738,740,743,746,748,751],{"class":75,"line":488},[208,736,737],{"class":250},"        df ",[208,739,280],{"class":246},[208,741,742],{"class":250}," df.drop(",[208,744,745],{"class":477},"columns",[208,747,280],{"class":246},[208,749,750],{"class":250},"unnamed)              ",[208,752,753],{"class":602},"# blank columns pandas auto-named\n",[208,755,756,758],{"class":75,"line":493},[208,757,471],{"class":246},[208,759,760],{"class":250}," df\n",[208,762,763],{"class":75,"line":509},[208,764,264],{"emptyLinePlaceholder":263},[208,766,768,770,772],{"class":75,"line":767},17,[208,769,496],{"class":250},[208,771,280],{"class":246},[208,773,774],{"class":250}," normalise_headers(df)\n",[208,776,778,780,782,785],{"class":75,"line":777},18,[208,779,512],{"class":368},[208,781,436],{"class":250},[208,783,784],{"class":368},"list",[208,786,787],{"class":250},"(df.columns))\n",[10,789,790,791,794],{},"Dropping the ",[14,792,793],{},"unnamed:"," columns is worth doing early. They appear whenever the sheet has a stray value or formatting to the right of the real table, and they otherwise trip every \"unexpected column\" check you write afterwards.",[194,796,798],{"id":797},"map-aliases-to-canonical-names","Map aliases to canonical names",[10,800,801,802,804,805,807],{},"A column that is ",[14,803,16],{}," in one submission and ",[14,806,20],{}," in the next is a naming problem, not a data problem. Keep the mapping in one dictionary:",[199,809,811],{"className":237,"code":810,"language":239,"meta":204,"style":204},"import pandas as pd\n\nALIASES = {\n    \"qty\": \"quantity\",\n    \"quantity_ordered\": \"quantity\",\n    \"unit_cost\": \"unit_price\",\n    \"price\": \"unit_price\",\n    \"order_no\": \"order_id\",\n    \"order_number\": \"order_id\",\n    \"date\": \"order_date\",\n}\n\ndef apply_aliases(df):\n    used = {c: ALIASES[c] for c in df.columns if c in ALIASES}\n    if used:\n        print(\"renamed:\", used)\n    return df.rename(columns=ALIASES)\n\ndf = apply_aliases(df)\n",[14,812,813,823,827,838,852,863,875,886,898,909,921,926,930,939,973,980,993,1008,1012],{"__ignoreMap":204},[208,814,815,817,819,821],{"class":75,"line":210},[208,816,247],{"class":246},[208,818,251],{"class":250},[208,820,254],{"class":246},[208,822,257],{"class":250},[208,824,825],{"class":75,"line":260},[208,826,264],{"emptyLinePlaceholder":263},[208,828,829,832,835],{"class":75,"line":267},[208,830,831],{"class":368},"ALIASES",[208,833,834],{"class":246}," =",[208,836,837],{"class":250}," {\n",[208,839,840,843,846,849],{"class":75,"line":289},[208,841,842],{"class":217},"    \"qty\"",[208,844,845],{"class":250},": ",[208,847,848],{"class":217},"\"quantity\"",[208,850,851],{"class":250},",\n",[208,853,854,857,859,861],{"class":75,"line":300},[208,855,856],{"class":217},"    \"quantity_ordered\"",[208,858,845],{"class":250},[208,860,848],{"class":217},[208,862,851],{"class":250},[208,864,865,868,870,873],{"class":75,"line":318},[208,866,867],{"class":217},"    \"unit_cost\"",[208,869,845],{"class":250},[208,871,872],{"class":217},"\"unit_price\"",[208,874,851],{"class":250},[208,876,877,880,882,884],{"class":75,"line":353},[208,878,879],{"class":217},"    \"price\"",[208,881,845],{"class":250},[208,883,872],{"class":217},[208,885,851],{"class":250},[208,887,888,891,893,896],{"class":75,"line":362},[208,889,890],{"class":217},"    \"order_no\"",[208,892,845],{"class":250},[208,894,895],{"class":217},"\"order_id\"",[208,897,851],{"class":250},[208,899,900,903,905,907],{"class":75,"line":375},[208,901,902],{"class":217},"    \"order_number\"",[208,904,845],{"class":250},[208,906,895],{"class":217},[208,908,851],{"class":250},[208,910,911,914,916,919],{"class":75,"line":416},[208,912,913],{"class":217},"    \"date\"",[208,915,845],{"class":250},[208,917,918],{"class":217},"\"order_date\"",[208,920,851],{"class":250},[208,922,923],{"class":75,"line":422},[208,924,925],{"class":250},"}\n",[208,927,928],{"class":75,"line":428},[208,929,264],{"emptyLinePlaceholder":263},[208,931,932,934,937],{"class":75,"line":468},[208,933,270],{"class":246},[208,935,936],{"class":273}," apply_aliases",[208,938,665],{"class":250},[208,940,941,944,946,949,951,954,956,958,960,962,964,966,968,971],{"class":75,"line":488},[208,942,943],{"class":250},"    used ",[208,945,280],{"class":246},[208,947,948],{"class":250}," {c: ",[208,950,831],{"class":368},[208,952,953],{"class":250},"[c] ",[208,955,329],{"class":246},[208,957,690],{"class":250},[208,959,335],{"class":246},[208,961,714],{"class":250},[208,963,341],{"class":246},[208,965,690],{"class":250},[208,967,335],{"class":246},[208,969,970],{"class":368}," ALIASES",[208,972,925],{"class":250},[208,974,975,977],{"class":75,"line":493},[208,976,303],{"class":246},[208,978,979],{"class":250}," used:\n",[208,981,982,985,987,990],{"class":75,"line":509},[208,983,984],{"class":368},"        print",[208,986,436],{"class":250},[208,988,989],{"class":217},"\"renamed:\"",[208,991,992],{"class":250},", used)\n",[208,994,995,997,1000,1002,1004,1006],{"class":75,"line":767},[208,996,471],{"class":246},[208,998,999],{"class":250}," df.rename(",[208,1001,745],{"class":477},[208,1003,280],{"class":246},[208,1005,831],{"class":368},[208,1007,465],{"class":250},[208,1009,1010],{"class":75,"line":777},[208,1011,264],{"emptyLinePlaceholder":263},[208,1013,1015,1017,1019],{"class":75,"line":1014},19,[208,1016,496],{"class":250},[208,1018,280],{"class":246},[208,1020,1021],{"class":250}," apply_aliases(df)\n",[10,1023,1024,1025,1028,1029,1032,1033,1036,1037,1039],{},"Logging which aliases fired matters more than it looks: when a report's numbers change unexpectedly, \"this month the file used ",[14,1026,1027],{},"price"," rather than ",[14,1030,1031],{},"unit_price","\" is often the whole explanation. Keeping the map explicit also stops the tempting fuzzy-matching approach, which eventually maps ",[14,1034,1035],{},"discount_price"," onto ",[14,1038,1031],{}," and produces a report nobody can reconcile.",[29,1041,38,1046,38,1049,38,1052,38,1055,38,1060,38,1063,38,1067,38,1074,38,1079,38,1084,38,1087,38,1090,38,1094,38,1098,38,1101,38,1105,38,1109,38,1112,38,1115,38,1118,38,1121,38,1123,38,1126,38,1128,38,1131,38,1133,38,1135,38,1139,38,1142,38,1145,38,1150],{"viewBox":1042,"role":32,"ariaLabelledBy":1043,"xmlns":36,"style":37},"0 0 740 224",[1044,1045],"vc-alias-t","vc-alias-d",[40,1047,1048],{"id":1044},"Header normalisation and alias mapping in two steps",[44,1050,1051],{"id":1045},"The raw header \" Order No \" becomes order_no after whitespace and case normalisation, then order_id after the alias map is applied. The raw header \"Unit Cost\" becomes unit_cost and then unit_price. Both end at canonical names the importer knows.",[48,1053],{"x":50,"y":50,"width":51,"height":1054,"fill":53},"224",[64,1056,1059],{"x":58,"y":1057,"style":1058},"34","font-size:12px;font-weight:700;fill:var(--muted,#5b6780);text-anchor:middle","as typed",[64,1061,1062],{"x":189,"y":1057,"style":1058},"normalised",[64,1064,1066],{"x":1065,"y":1057,"style":1058},"620","canonical",[48,1068],{"x":1069,"y":1070,"width":1071,"height":190,"rx":1072,"fill":1073,"stroke":62},"24","50","192","8","#f0f2f5",[64,1075,1078],{"x":58,"y":1076,"style":1077},"76","font-size:12.5px;fill:var(--text,#172033);text-anchor:middle;font-family:ui-monospace,Menlo,monospace","\" Order No \"",[75,1080],{"x1":1081,"y1":1082,"x2":1083,"y2":1082,"stroke":80,"style":180},"220","71","266",[83,1085],{"points":1086,"fill":86},"270,71 258,65 258,77",[48,1088],{"x":1089,"y":1070,"width":1071,"height":190,"rx":1072,"fill":61,"stroke":62},"274",[64,1091,1093],{"x":189,"y":1076,"style":1092},"font-size:12.5px;fill:var(--brand-strong,#4338ca);text-anchor:middle;font-family:ui-monospace,Menlo,monospace","order_no",[75,1095],{"x1":1096,"y1":1082,"x2":1097,"y2":1082,"stroke":153,"style":180},"470","516",[83,1099],{"points":1100,"fill":157},"520,71 508,65 508,77",[48,1102],{"x":1103,"y":1070,"width":1071,"height":190,"rx":1072,"fill":1104,"stroke":153},"524","#d9f4f1",[64,1106,1108],{"x":1065,"y":1076,"style":1107},"font-size:12.5px;font-weight:700;fill:var(--teal-ink,#0b6157);text-anchor:middle;font-family:ui-monospace,Menlo,monospace","order_id",[48,1110],{"x":1069,"y":1111,"width":1071,"height":190,"rx":1072,"fill":1073,"stroke":62},"106",[64,1113,1114],{"x":58,"y":72,"style":1077},"\"Unit Cost\"",[75,1116],{"x1":1081,"y1":1117,"x2":1083,"y2":1117,"stroke":80,"style":180},"127",[83,1119],{"points":1120,"fill":86},"270,127 258,121 258,133",[48,1122],{"x":1089,"y":1111,"width":1071,"height":190,"rx":1072,"fill":61,"stroke":62},[64,1124,1125],{"x":189,"y":72,"style":1092},"unit_cost",[75,1127],{"x1":1096,"y1":1117,"x2":1097,"y2":1117,"stroke":153,"style":180},[83,1129],{"points":1130,"fill":157},"520,127 508,121 508,133",[48,1132],{"x":1103,"y":1111,"width":1071,"height":190,"rx":1072,"fill":1104,"stroke":153},[64,1134,1031],{"x":1065,"y":72,"style":1107},[64,1136,1138],{"x":58,"y":179,"style":1137},"font-size:11.5px;fill:var(--muted,#5b6780);text-anchor:middle","strip, collapse, lower",[64,1140,1141],{"x":189,"y":179,"style":1137},"one deterministic form",[64,1143,1144],{"x":1065,"y":179,"style":1137},"explicit alias map",[48,1146],{"x":1147,"y":1071,"width":1148,"height":1069,"rx":1072,"fill":1149},"180","380","#fdefd8",[64,1151,1154],{"x":189,"y":1152,"style":1153},"209","font-size:11.5px;fill:var(--gold,#b4740a);text-anchor:middle","log every alias that fires — it explains changed numbers later",[194,1156,1158],{"id":1157},"assert-the-contract","Assert the contract",[10,1160,1161],{},"With headers canonical, the actual check is small — and its message should say what to do:",[199,1163,1165],{"className":237,"code":1164,"language":239,"meta":204,"style":204},"import pandas as pd\n\nREQUIRED = {\"order_id\", \"region\", \"order_date\", \"quantity\", \"unit_price\"}\nOPTIONAL = {\"notes\", \"sales_rep\"}\n\ndef check_columns(df, path):\n    present = set(df.columns)\n    missing = REQUIRED - present\n    unexpected = present - REQUIRED - OPTIONAL\n\n    problems = []\n    if missing:\n        problems.append(f\"missing required column(s): {sorted(missing)}\")\n    if unexpected:\n        problems.append(f\"unexpected column(s), ignored: {sorted(unexpected)}\")\n    if df.empty:\n        problems.append(\"no data rows below the header\")\n\n    if missing or df.empty:\n        raise ValueError(f\"{path} failed structural validation — \" + \"; \".join(problems))\n    for note in problems:\n        print(\"warning:\", note)\n\n    return df[sorted(REQUIRED | (present & OPTIONAL))]\n\ndf = check_columns(df, \"submitted.xlsx\")\n",[14,1166,1167,1177,1181,1212,1231,1235,1245,1258,1274,1294,1298,1308,1315,1339,1346,1368,1375,1384,1388,1400,1432,1446,1459,1464,1493,1498],{"__ignoreMap":204},[208,1168,1169,1171,1173,1175],{"class":75,"line":210},[208,1170,247],{"class":246},[208,1172,251],{"class":250},[208,1174,254],{"class":246},[208,1176,257],{"class":250},[208,1178,1179],{"class":75,"line":260},[208,1180,264],{"emptyLinePlaceholder":263},[208,1182,1183,1186,1188,1191,1193,1195,1198,1200,1202,1204,1206,1208,1210],{"class":75,"line":267},[208,1184,1185],{"class":368},"REQUIRED",[208,1187,834],{"class":246},[208,1189,1190],{"class":250}," {",[208,1192,895],{"class":217},[208,1194,593],{"class":250},[208,1196,1197],{"class":217},"\"region\"",[208,1199,593],{"class":250},[208,1201,918],{"class":217},[208,1203,593],{"class":250},[208,1205,848],{"class":217},[208,1207,593],{"class":250},[208,1209,872],{"class":217},[208,1211,925],{"class":250},[208,1213,1214,1217,1219,1221,1224,1226,1229],{"class":75,"line":289},[208,1215,1216],{"class":368},"OPTIONAL",[208,1218,834],{"class":246},[208,1220,1190],{"class":250},[208,1222,1223],{"class":217},"\"notes\"",[208,1225,593],{"class":250},[208,1227,1228],{"class":217},"\"sales_rep\"",[208,1230,925],{"class":250},[208,1232,1233],{"class":75,"line":300},[208,1234,264],{"emptyLinePlaceholder":263},[208,1236,1237,1239,1242],{"class":75,"line":318},[208,1238,270],{"class":246},[208,1240,1241],{"class":273}," check_columns",[208,1243,1244],{"class":250},"(df, path):\n",[208,1246,1247,1250,1252,1255],{"class":75,"line":353},[208,1248,1249],{"class":250},"    present ",[208,1251,280],{"class":246},[208,1253,1254],{"class":368}," set",[208,1256,1257],{"class":250},"(df.columns)\n",[208,1259,1260,1263,1265,1268,1271],{"class":75,"line":362},[208,1261,1262],{"class":250},"    missing ",[208,1264,280],{"class":246},[208,1266,1267],{"class":368}," REQUIRED",[208,1269,1270],{"class":246}," -",[208,1272,1273],{"class":250}," present\n",[208,1275,1276,1279,1281,1284,1287,1289,1291],{"class":75,"line":375},[208,1277,1278],{"class":250},"    unexpected ",[208,1280,280],{"class":246},[208,1282,1283],{"class":250}," present ",[208,1285,1286],{"class":246},"-",[208,1288,1267],{"class":368},[208,1290,1270],{"class":246},[208,1292,1293],{"class":368}," OPTIONAL\n",[208,1295,1296],{"class":75,"line":416},[208,1297,264],{"emptyLinePlaceholder":263},[208,1299,1300,1303,1305],{"class":75,"line":422},[208,1301,1302],{"class":250},"    problems ",[208,1304,280],{"class":246},[208,1306,1307],{"class":250}," []\n",[208,1309,1310,1312],{"class":75,"line":428},[208,1311,303],{"class":246},[208,1313,1314],{"class":250}," missing:\n",[208,1316,1317,1320,1322,1325,1327,1330,1333,1335,1337],{"class":75,"line":468},[208,1318,1319],{"class":250},"        problems.append(",[208,1321,439],{"class":246},[208,1323,1324],{"class":217},"\"missing required column(s): ",[208,1326,385],{"class":384},[208,1328,1329],{"class":368},"sorted",[208,1331,1332],{"class":250},"(missing)",[208,1334,393],{"class":384},[208,1336,462],{"class":217},[208,1338,465],{"class":250},[208,1340,1341,1343],{"class":75,"line":488},[208,1342,303],{"class":246},[208,1344,1345],{"class":250}," unexpected:\n",[208,1347,1348,1350,1352,1355,1357,1359,1362,1364,1366],{"class":75,"line":493},[208,1349,1319],{"class":250},[208,1351,439],{"class":246},[208,1353,1354],{"class":217},"\"unexpected column(s), ignored: ",[208,1356,385],{"class":384},[208,1358,1329],{"class":368},[208,1360,1361],{"class":250},"(unexpected)",[208,1363,393],{"class":384},[208,1365,462],{"class":217},[208,1367,465],{"class":250},[208,1369,1370,1372],{"class":75,"line":509},[208,1371,303],{"class":246},[208,1373,1374],{"class":250}," df.empty:\n",[208,1376,1377,1379,1382],{"class":75,"line":767},[208,1378,1319],{"class":250},[208,1380,1381],{"class":217},"\"no data rows below the header\"",[208,1383,465],{"class":250},[208,1385,1386],{"class":75,"line":777},[208,1387,264],{"emptyLinePlaceholder":263},[208,1389,1390,1392,1395,1398],{"class":75,"line":1014},[208,1391,303],{"class":246},[208,1393,1394],{"class":250}," missing ",[208,1396,1397],{"class":246},"or",[208,1399,1374],{"class":250},[208,1401,1403,1405,1407,1409,1411,1413,1415,1418,1420,1423,1426,1429],{"class":75,"line":1402},20,[208,1404,431],{"class":246},[208,1406,369],{"class":368},[208,1408,436],{"class":250},[208,1410,439],{"class":246},[208,1412,462],{"class":217},[208,1414,385],{"class":384},[208,1416,1417],{"class":250},"path",[208,1419,393],{"class":384},[208,1421,1422],{"class":217}," failed structural validation — \"",[208,1424,1425],{"class":246}," +",[208,1427,1428],{"class":217}," \"; \"",[208,1430,1431],{"class":250},".join(problems))\n",[208,1433,1435,1438,1441,1443],{"class":75,"line":1434},21,[208,1436,1437],{"class":246},"    for",[208,1439,1440],{"class":250}," note ",[208,1442,335],{"class":246},[208,1444,1445],{"class":250}," problems:\n",[208,1447,1449,1451,1453,1456],{"class":75,"line":1448},22,[208,1450,984],{"class":368},[208,1452,436],{"class":250},[208,1454,1455],{"class":217},"\"warning:\"",[208,1457,1458],{"class":250},", note)\n",[208,1460,1462],{"class":75,"line":1461},23,[208,1463,264],{"emptyLinePlaceholder":263},[208,1465,1467,1469,1472,1474,1476,1478,1481,1484,1487,1490],{"class":75,"line":1466},24,[208,1468,471],{"class":246},[208,1470,1471],{"class":250}," df[",[208,1473,1329],{"class":368},[208,1475,436],{"class":250},[208,1477,1185],{"class":368},[208,1479,1480],{"class":246}," |",[208,1482,1483],{"class":250}," (present ",[208,1485,1486],{"class":246},"&",[208,1488,1489],{"class":368}," OPTIONAL",[208,1491,1492],{"class":250},"))]\n",[208,1494,1496],{"class":75,"line":1495},25,[208,1497,264],{"emptyLinePlaceholder":263},[208,1499,1501,1503,1505,1508,1510],{"class":75,"line":1500},26,[208,1502,496],{"class":250},[208,1504,280],{"class":246},[208,1506,1507],{"class":250}," check_columns(df, ",[208,1509,504],{"class":217},[208,1511,465],{"class":250},[10,1513,1514,1515,1518],{},"Note the asymmetry: missing columns are fatal, unexpected ones are a warning. A submitter who adds a ",[14,1516,1517],{},"comments"," column has not broken anything, and rejecting their file teaches them that the process is brittle. Selecting the columns explicitly at the end also fixes the order for everything downstream, so no later code can depend on the submitter's layout.",[194,1520,1522],{"id":1521},"when-the-header-row-is-not-row-1","When the header row is not row 1",[10,1524,1525,1526,1529],{},"Exports often carry a title and a timestamp above the real header. Find the header rather than hardcoding ",[14,1527,1528],{},"skiprows",", so a file with one extra line does not fail:",[199,1531,1533],{"className":237,"code":1532,"language":239,"meta":204,"style":204},"import pandas as pd\n\ndef find_header_row(path, sheet, required, max_scan=10):\n    probe = pd.read_excel(path, sheet_name=sheet, header=None, nrows=max_scan, dtype=object)\n    for i, row in probe.iterrows():\n        cells = {normalise(v) for v in row if pd.notna(v)}\n        if required \u003C= cells:\n            return i\n    raise ValueError(\n        f\"no header row containing {sorted(required)} found in the first {max_scan} rows\"\n    )\n\nheader_row = find_header_row(\"export.xlsx\", \"Export\", {\"order_id\", \"quantity\"})\ndf = pd.read_excel(\"export.xlsx\", sheet_name=\"Export\", header=header_row, dtype=object)\nprint(\"header found on sheet row\", header_row + 1)\n",[14,1534,1535,1545,1549,1565,1609,1621,1646,1659,1667,1676,1706,1711,1715,1745,1781],{"__ignoreMap":204},[208,1536,1537,1539,1541,1543],{"class":75,"line":210},[208,1538,247],{"class":246},[208,1540,251],{"class":250},[208,1542,254],{"class":246},[208,1544,257],{"class":250},[208,1546,1547],{"class":75,"line":260},[208,1548,264],{"emptyLinePlaceholder":263},[208,1550,1551,1553,1556,1559,1561,1563],{"class":75,"line":267},[208,1552,270],{"class":246},[208,1554,1555],{"class":273}," find_header_row",[208,1557,1558],{"class":250},"(path, sheet, required, max_scan",[208,1560,280],{"class":246},[208,1562,93],{"class":368},[208,1564,286],{"class":250},[208,1566,1567,1570,1572,1575,1578,1580,1583,1586,1588,1591,1593,1596,1598,1601,1603,1605,1607],{"class":75,"line":289},[208,1568,1569],{"class":250},"    probe ",[208,1571,280],{"class":246},[208,1573,1574],{"class":250}," pd.read_excel(path, ",[208,1576,1577],{"class":477},"sheet_name",[208,1579,280],{"class":246},[208,1581,1582],{"class":250},"sheet, ",[208,1584,1585],{"class":477},"header",[208,1587,280],{"class":246},[208,1589,1590],{"class":368},"None",[208,1592,593],{"class":250},[208,1594,1595],{"class":477},"nrows",[208,1597,280],{"class":246},[208,1599,1600],{"class":250},"max_scan, ",[208,1602,478],{"class":477},[208,1604,280],{"class":246},[208,1606,483],{"class":368},[208,1608,465],{"class":250},[208,1610,1611,1613,1616,1618],{"class":75,"line":300},[208,1612,1437],{"class":246},[208,1614,1615],{"class":250}," i, row ",[208,1617,335],{"class":246},[208,1619,1620],{"class":250}," probe.iterrows():\n",[208,1622,1623,1626,1628,1631,1633,1636,1638,1641,1643],{"class":75,"line":318},[208,1624,1625],{"class":250},"        cells ",[208,1627,280],{"class":246},[208,1629,1630],{"class":250}," {normalise(v) ",[208,1632,329],{"class":246},[208,1634,1635],{"class":250}," v ",[208,1637,335],{"class":246},[208,1639,1640],{"class":250}," row ",[208,1642,341],{"class":246},[208,1644,1645],{"class":250}," pd.notna(v)}\n",[208,1647,1648,1650,1653,1656],{"class":75,"line":353},[208,1649,356],{"class":246},[208,1651,1652],{"class":250}," required ",[208,1654,1655],{"class":246},"\u003C=",[208,1657,1658],{"class":250}," cells:\n",[208,1660,1661,1664],{"class":75,"line":362},[208,1662,1663],{"class":246},"            return",[208,1665,1666],{"class":250}," i\n",[208,1668,1669,1672,1674],{"class":75,"line":375},[208,1670,1671],{"class":246},"    raise",[208,1673,369],{"class":368},[208,1675,372],{"class":250},[208,1677,1678,1681,1684,1686,1688,1691,1693,1696,1698,1701,1703],{"class":75,"line":416},[208,1679,1680],{"class":246},"        f",[208,1682,1683],{"class":217},"\"no header row containing ",[208,1685,385],{"class":384},[208,1687,1329],{"class":368},[208,1689,1690],{"class":250},"(required)",[208,1692,393],{"class":384},[208,1694,1695],{"class":217}," found in the first ",[208,1697,385],{"class":384},[208,1699,1700],{"class":250},"max_scan",[208,1702,393],{"class":384},[208,1704,1705],{"class":217}," rows\"\n",[208,1707,1708],{"class":75,"line":422},[208,1709,1710],{"class":250},"    )\n",[208,1712,1713],{"class":75,"line":428},[208,1714,264],{"emptyLinePlaceholder":263},[208,1716,1717,1720,1722,1725,1728,1730,1733,1736,1738,1740,1742],{"class":75,"line":468},[208,1718,1719],{"class":250},"header_row ",[208,1721,280],{"class":246},[208,1723,1724],{"class":250}," find_header_row(",[208,1726,1727],{"class":217},"\"export.xlsx\"",[208,1729,593],{"class":250},[208,1731,1732],{"class":217},"\"Export\"",[208,1734,1735],{"class":250},", {",[208,1737,895],{"class":217},[208,1739,593],{"class":250},[208,1741,848],{"class":217},[208,1743,1744],{"class":250},"})\n",[208,1746,1747,1749,1751,1754,1756,1758,1760,1762,1764,1766,1768,1770,1773,1775,1777,1779],{"class":75,"line":488},[208,1748,496],{"class":250},[208,1750,280],{"class":246},[208,1752,1753],{"class":250}," pd.read_excel(",[208,1755,1727],{"class":217},[208,1757,593],{"class":250},[208,1759,1577],{"class":477},[208,1761,280],{"class":246},[208,1763,1732],{"class":217},[208,1765,593],{"class":250},[208,1767,1585],{"class":477},[208,1769,280],{"class":246},[208,1771,1772],{"class":250},"header_row, ",[208,1774,478],{"class":477},[208,1776,280],{"class":246},[208,1778,483],{"class":368},[208,1780,465],{"class":250},[208,1782,1783,1785,1787,1790,1793,1795,1798],{"class":75,"line":493},[208,1784,512],{"class":368},[208,1786,436],{"class":250},[208,1788,1789],{"class":217},"\"header found on sheet row\"",[208,1791,1792],{"class":250},", header_row ",[208,1794,623],{"class":246},[208,1796,1797],{"class":368}," 1",[208,1799,465],{"class":250},[10,1801,1802,1803,1807,1808,1810],{},"This is the same problem ",[23,1804,1806],{"href":1805},"\u002Fgetting-started-with-python-excel-automation\u002Freading-excel-files-with-pandas\u002F","reading Excel files with pandas"," solves with ",[14,1809,1528],{},", made adaptive: instead of asserting that the junk is two rows deep, you assert what the header must contain.",[194,1812,1814],{"id":1813},"putting-the-gate-together","Putting the gate together",[199,1816,1818],{"className":237,"code":1817,"language":239,"meta":204,"style":204},"def load_validated(path, sheet=\"Orders\"):\n    df = open_sheet(path, sheet)\n    df = normalise_headers(df)\n    df = apply_aliases(df)\n    return check_columns(df, path)\n\ntry:\n    orders = load_validated(\"submitted.xlsx\")\nexcept ValueError as exc:\n    raise SystemExit(f\"import rejected: {exc}\")\n\nprint(f\"structure OK — {len(orders)} row(s), columns {list(orders.columns)}\")\n",[14,1819,1820,1835,1844,1852,1860,1867,1871,1879,1893,1906,1931,1935],{"__ignoreMap":204},[208,1821,1822,1824,1827,1829,1831,1833],{"class":75,"line":210},[208,1823,270],{"class":246},[208,1825,1826],{"class":273}," load_validated",[208,1828,277],{"class":250},[208,1830,280],{"class":246},[208,1832,283],{"class":217},[208,1834,286],{"class":250},[208,1836,1837,1839,1841],{"class":75,"line":260},[208,1838,670],{"class":250},[208,1840,280],{"class":246},[208,1842,1843],{"class":250}," open_sheet(path, sheet)\n",[208,1845,1846,1848,1850],{"class":75,"line":267},[208,1847,670],{"class":250},[208,1849,280],{"class":246},[208,1851,774],{"class":250},[208,1853,1854,1856,1858],{"class":75,"line":289},[208,1855,670],{"class":250},[208,1857,280],{"class":246},[208,1859,1021],{"class":250},[208,1861,1862,1864],{"class":75,"line":300},[208,1863,471],{"class":246},[208,1865,1866],{"class":250}," check_columns(df, path)\n",[208,1868,1869],{"class":75,"line":318},[208,1870,264],{"emptyLinePlaceholder":263},[208,1872,1873,1876],{"class":75,"line":353},[208,1874,1875],{"class":246},"try",[208,1877,1878],{"class":250},":\n",[208,1880,1881,1884,1886,1889,1891],{"class":75,"line":362},[208,1882,1883],{"class":250},"    orders ",[208,1885,280],{"class":246},[208,1887,1888],{"class":250}," load_validated(",[208,1890,504],{"class":217},[208,1892,465],{"class":250},[208,1894,1895,1898,1900,1903],{"class":75,"line":375},[208,1896,1897],{"class":246},"except",[208,1899,369],{"class":368},[208,1901,1902],{"class":246}," as",[208,1904,1905],{"class":250}," exc:\n",[208,1907,1908,1910,1913,1915,1917,1920,1922,1925,1927,1929],{"class":75,"line":416},[208,1909,1671],{"class":246},[208,1911,1912],{"class":368}," SystemExit",[208,1914,436],{"class":250},[208,1916,439],{"class":246},[208,1918,1919],{"class":217},"\"import rejected: ",[208,1921,385],{"class":384},[208,1923,1924],{"class":250},"exc",[208,1926,393],{"class":384},[208,1928,462],{"class":217},[208,1930,465],{"class":250},[208,1932,1933],{"class":75,"line":422},[208,1934,264],{"emptyLinePlaceholder":263},[208,1936,1937,1939,1941,1943,1946,1948,1951,1954,1956,1959,1961,1963,1966,1968,1970],{"class":75,"line":428},[208,1938,512],{"class":368},[208,1940,436],{"class":250},[208,1942,439],{"class":246},[208,1944,1945],{"class":217},"\"structure OK — ",[208,1947,385],{"class":384},[208,1949,1950],{"class":368},"len",[208,1952,1953],{"class":250},"(orders)",[208,1955,393],{"class":384},[208,1957,1958],{"class":217}," row(s), columns ",[208,1960,385],{"class":384},[208,1962,784],{"class":368},[208,1964,1965],{"class":250},"(orders.columns)",[208,1967,393],{"class":384},[208,1969,462],{"class":217},[208,1971,465],{"class":250},[10,1973,1974],{},"Four small functions, each with one job, and a single entry point the rest of the pipeline calls. Row-level checks — types, ranges, duplicates — run after this gate, on data whose shape is already guaranteed.",[194,1976,1978],{"id":1977},"fatal-warning-or-silent","Fatal, warning or silent?",[10,1980,1981],{},"Not every structural surprise deserves the same response, and deciding once saves an argument every month:",[29,1983,38,1988,38,1991,38,1994,38,1997,38,2004,38,2010,38,2014,38,2018,38,2022,38,2025,38,2029,38,2033,38,2036,38,2039,38,2042,38,2045,38,2047,38,2052,38,2055,38,2058,38,2061],{"viewBox":1984,"role":32,"ariaLabelledBy":1985,"xmlns":36,"style":37},"0 0 740 216",[1986,1987],"vc-sev-t","vc-sev-d",[40,1989,1990],{"id":1986},"How to respond to each kind of structural surprise",[44,1992,1993],{"id":1987},"Missing sheet, missing required column and an empty sheet are fatal. An unexpected extra column and a fired alias are warnings that the run logs. Column order and trailing blank columns are handled silently.",[48,1995],{"x":50,"y":50,"width":51,"height":1996,"fill":53},"216",[48,1998],{"x":1999,"y":2000,"width":2001,"height":2002,"rx":2003,"fill":168,"stroke":169,"style":81},"16","26","228","170","14",[64,2005,2009],{"x":2006,"y":2007,"style":2008},"130","54","font-size:13px;font-weight:700;fill:var(--danger,#dc2626);text-anchor:middle","fatal — stop",[64,2011,2013],{"x":2006,"y":57,"style":2012},"font-size:11.5px;fill:var(--text,#172033);text-anchor:middle","sheet not found",[64,2015,2017],{"x":2006,"y":2016,"style":2012},"110","required column missing",[64,2019,2021],{"x":2006,"y":2020,"style":2012},"134","no data rows",[64,2023,2024],{"x":2006,"y":2002,"style":104},"nothing downstream is safe",[48,2026],{"x":2027,"y":2000,"width":2001,"height":2002,"rx":2003,"fill":1149,"stroke":2028,"style":81},"256","var(--gold,#b4740a)",[64,2030,2032],{"x":189,"y":2007,"style":2031},"font-size:13px;font-weight:700;fill:var(--gold-ink,#7a4e06);text-anchor:middle","warn — log and continue",[64,2034,2035],{"x":189,"y":57,"style":2012},"unexpected extra column",[64,2037,2038],{"x":189,"y":2016,"style":2012},"an alias had to fire",[64,2040,2041],{"x":189,"y":2020,"style":2012},"header found below row 1",[64,2043,2044],{"x":189,"y":2002,"style":104},"explains changes later",[48,2046],{"x":166,"y":2000,"width":2001,"height":2002,"rx":2003,"fill":1104,"stroke":153,"style":81},[64,2048,2051],{"x":2049,"y":2007,"style":2050},"610","font-size:13px;font-weight:700;fill:var(--teal-ink,#0b6157);text-anchor:middle","silent — just handle it",[64,2053,2054],{"x":2049,"y":57,"style":2012},"different column order",[64,2056,2057],{"x":2049,"y":2016,"style":2012},"trailing blank columns",[64,2059,2060],{"x":2049,"y":2020,"style":2012},"case and spacing in headers",[64,2062,2063],{"x":2049,"y":2002,"style":104},"not the submitter's problem",[10,2065,2066],{},"The right-hand column is the one teams get wrong most often, usually by being too strict. Rejecting a file because the submitter reordered two columns or left a blank column at the edge makes the import look fragile and trains people to work around it. Absorb everything that carries no information, warn about everything that might explain a future discrepancy, and stop only when the file genuinely cannot be processed.",[194,2068,2070],{"id":2069},"common-pitfalls-and-gotchas","Common pitfalls and gotchas",[2072,2073,2074,2090],"table",{},[2075,2076,2077],"thead",{},[2078,2079,2080,2084,2087],"tr",{},[2081,2082,2083],"th",{},"Symptom",[2081,2085,2086],{},"Cause",[2081,2088,2089],{},"Fix",[2091,2092,2093,2108,2122,2138,2149,2160],"tbody",{},[2078,2094,2095,2102,2105],{},[2096,2097,2098,2101],"td",{},[14,2099,2100],{},"KeyError"," deep inside the transform",[2096,2103,2104],{},"Column checked too late, or not at all",[2096,2106,2107],{},"Run the structural gate before anything else",[2078,2109,2110,2113,2116],{},[2096,2111,2112],{},"Header comparison fails on identical-looking text",[2096,2114,2115],{},"Non-breaking spaces or newlines in the header cell",[2096,2117,2118,2119,2121],{},"Normalise with ",[14,2120,588],{}," replacement and whitespace collapsing",[2078,2123,2124,2130,2133],{},[2096,2125,2126,2129],{},[14,2127,2128],{},"Unnamed: 7"," columns appear",[2096,2131,2132],{},"Stray value or formatting right of the table",[2096,2134,2135,2136],{},"Drop columns whose normalised name starts with ",[14,2137,793],{},[2078,2139,2140,2143,2146],{},[2096,2141,2142],{},"Import breaks when a column is added",[2096,2144,2145],{},"Code selected columns by position",[2096,2147,2148],{},"Select by name after validating",[2078,2150,2151,2154,2157],{},[2096,2152,2153],{},"Empty report, no error",[2096,2155,2156],{},"Sheet had headers but no rows",[2096,2158,2159],{},"Treat an empty frame as a structural failure",[2078,2161,2162,2165,2171],{},[2096,2163,2164],{},"Wrong sheet silently read",[2096,2166,2167,2170],{},[14,2168,2169],{},"sheet_name=0"," picked whatever was first",[2096,2172,2173],{},"Name the sheet and verify it exists",[194,2175,2177],{"id":2176},"performance-and-scale-notes","Performance and scale notes",[10,2179,2180,2183,2184,2187,2188,2191],{},[14,2181,2182],{},"pd.ExcelFile(path)"," parses the workbook structure without materialising rows, so the sheet check is cheap even on a large file. Header discovery with ",[14,2185,2186],{},"nrows=10"," reads only the top of the sheet. The expensive step is the full read, which is why the gate is arranged to fail before it: on a 200 MB submission with the wrong tab name, this ordering is the difference between a one-second rejection and a two-minute one. When you only need a subset of columns, pass ",[14,2189,2190],{},"usecols"," on the real read once the names are known, and keep the gate itself free of row-level work so its cost stays proportional to the header rather than to the data.",[194,2193,2195],{"id":2194},"conclusion","Conclusion",[10,2197,2198],{},"A structural gate is four questions asked in order: is the sheet there, do the headers normalise to names I know, are the required ones present, and is there any data. Answer them before reading rows, keep aliases in an explicit map, make missing columns fatal and unexpected ones a warning, and every downstream stage can assume a known shape.",[194,2200,2202],{"id":2201},"frequently-asked-questions","Frequently asked questions",[10,2204,2205,2209],{},[2206,2207,2208],"strong",{},"Should column order matter?","\nUsually not. Validate by name and select the columns you need in your own order, so an extra column inserted at position two cannot break the import.",[10,2211,2212,2215],{},[2206,2213,2214],{},"How do I handle a column that gets renamed every quarter?","\nKeep an alias map from every known spelling to your canonical name, apply it after normalising case and whitespace, and log which alias was used.",[10,2217,2218,2221],{},[2206,2219,2220],{},"Is it better to fail or to fill in a missing column?","\nFail. A silently added empty column produces a report full of zeros that nobody questions, which is worse than an import that stops with a clear message.",[10,2223,2224,2227,2228,2230],{},[2206,2225,2226],{},"How do I check the header row when it is not the first row?","\nFind it by scanning the first few rows for your required names, then re-read the sheet with ",[14,2229,1585],{}," set to that row index.",[194,2232,2234],{"id":2233},"related","Related",[10,2236,2237],{},"Up to the parent guide:",[2239,2240,2241],"ul",{},[2242,2243,2244,2246],"li",{},[23,2245,26],{"href":25}," — where this gate sits in the wider validation layer.",[10,2248,2249],{},"Related guides:",[2239,2251,2252,2259,2266,2273],{},[2242,2253,2254,2258],{},[23,2255,2257],{"href":2256},"\u002Fadvanced-data-transformation-and-cleaning\u002Fvalidating-excel-data-with-python\u002Fcheck-excel-data-types-with-pandas\u002F","Check Excel Data Types with pandas"," — the row-level checks that run next.",[2242,2260,2261,2265],{},[23,2262,2264],{"href":2263},"\u002Fadvanced-data-transformation-and-cleaning\u002Fvalidating-excel-data-with-python\u002Ffind-duplicate-rows-in-excel-with-python\u002F","Find Duplicate Rows in Excel with Python"," — cross-row validation after the shape is known.",[2242,2267,2268,2272],{},[23,2269,2271],{"href":2270},"\u002Fgetting-started-with-python-excel-automation\u002Freading-excel-files-with-pandas\u002Fhow-to-read-excel-with-pandas-step-by-step\u002F","How to Read Excel with pandas, Step by Step"," — sheets, headers and skiprows in depth.",[2242,2274,2275,2279,2280,2282],{},[23,2276,2278],{"href":2277},"\u002Fgetting-started-with-python-excel-automation\u002Freading-excel-files-with-pandas\u002Fread-specific-columns-from-excel-with-pandas\u002F","Read Specific Columns from Excel with pandas"," — ",[14,2281,2190],{}," once the names are trusted.",[2284,2285,2286],"style",{},"html pre.shiki code .sMTad, html code.shiki .sMTad{--shiki-default:#6F42C1;--shiki-dark:#FFB757}html pre.shiki code .srMev, html code.shiki .srMev{--shiki-default:#032F62;--shiki-dark:#ADDCFF}html .default .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .dark .shiki span {color: var(--shiki-dark);background: var(--shiki-dark-bg);font-style: var(--shiki-dark-font-style);font-weight: var(--shiki-dark-font-weight);text-decoration: var(--shiki-dark-text-decoration);}html.dark .shiki span {color: var(--shiki-dark);background: var(--shiki-dark-bg);font-style: var(--shiki-dark-font-style);font-weight: var(--shiki-dark-font-weight);text-decoration: var(--shiki-dark-text-decoration);}html pre.shiki code .s-kum, html code.shiki .s-kum{--shiki-default:#D73A49;--shiki-dark:#FF9492}html pre.shiki code .skGVy, html code.shiki .skGVy{--shiki-default:#24292E;--shiki-dark:#F0F3F6}html pre.shiki code .s_Opv, html code.shiki .s_Opv{--shiki-default:#6F42C1;--shiki-dark:#DBB7FF}html pre.shiki code .sP0c6, html code.shiki .sP0c6{--shiki-default:#005CC5;--shiki-dark:#91CBFF}html pre.shiki code .sSjpA, html code.shiki .sSjpA{--shiki-default:#005CC5;--shiki-dark:#FF9492}html pre.shiki code .sa561, html code.shiki .sa561{--shiki-default:#E36209;--shiki-dark:#FFB757}html pre.shiki code .s-wDw, html code.shiki .s-wDw{--shiki-default:#6A737D;--shiki-dark:#BDC4CC}",{"title":204,"searchDepth":260,"depth":260,"links":2288},[2289,2290,2291,2292,2293,2294,2295,2296,2297,2298,2299,2300,2301],{"id":196,"depth":260,"text":197},{"id":227,"depth":260,"text":228},{"id":530,"depth":260,"text":531},{"id":797,"depth":260,"text":798},{"id":1157,"depth":260,"text":1158},{"id":1521,"depth":260,"text":1522},{"id":1813,"depth":260,"text":1814},{"id":1977,"depth":260,"text":1978},{"id":2069,"depth":260,"text":2070},{"id":2176,"depth":260,"text":2177},{"id":2194,"depth":260,"text":2195},{"id":2201,"depth":260,"text":2202},{"id":2233,"depth":260,"text":2234},"2026-08-01","Check an incoming workbook's sheet, headers, column order and required fields with pandas before a single row reaches your pipeline — with clear, actionable failure messages.","md",[2306,2308,2310,2312],{"q":2208,"a":2307},"Usually not. Validate by name and select the columns you need in your own order, so an extra column inserted at position two cannot break the import.",{"q":2214,"a":2309},"Keep an alias map from every known spelling to your canonical name, apply it after normalising case and whitespace, and log which alias was used.",{"q":2220,"a":2311},"Fail. A silently added empty column produces a report full of zeros that nobody questions, which is worse than an import that stops with a clear message.",{"q":2226,"a":2313},"Find it by scanning the first few rows for your required names, then re-read the sheet with header set to that row index.",{"breadcrumb":2315},[2316,2319,2322,2323],{"name":2317,"item":2318},"Home","\u002F",{"name":2320,"item":2321},"Advanced Data Transformation and Cleaning","\u002Fadvanced-data-transformation-and-cleaning\u002F",{"name":26,"item":25},{"name":5,"item":2324},"\u002Fadvanced-data-transformation-and-cleaning\u002Fvalidating-excel-data-with-python\u002Fvalidate-excel-columns-before-import-with-pandas\u002F","\u002Fadvanced-data-transformation-and-cleaning\u002Fvalidating-excel-data-with-python\u002Fvalidate-excel-columns-before-import-with-pandas",{"title":5,"description":2327},"A structural gate for Excel imports: verify the sheet exists, normalise headers, detect missing and unexpected columns, handle renames, and fail with a useful message.","validate-excel-columns-before-import-with-pandas","advanced-data-transformation-and-cleaning\u002Fvalidating-excel-data-with-python\u002Fvalidate-excel-columns-before-import-with-pandas\u002Findex","how-to","bC9i3OO8s9CGTChs-yb4ZbfKT3yW6W_5zhuSZbXFMTo",[2333,2337],{"title":2334,"path":2335,"stem":2336,"children":-1},"Highlight Invalid Cells in Excel with Python","\u002Fadvanced-data-transformation-and-cleaning\u002Fvalidating-excel-data-with-python\u002Fhighlight-invalid-cells-in-excel-with-python","advanced-data-transformation-and-cleaning\u002Fvalidating-excel-data-with-python\u002Fhighlight-invalid-cells-in-excel-with-python\u002Findex",{"title":2338,"path":2339,"stem":2340,"children":-1},"Working with Large Excel Files in Python","\u002Fadvanced-data-transformation-and-cleaning\u002Fworking-with-large-excel-files-in-python","advanced-data-transformation-and-cleaning\u002Fworking-with-large-excel-files-in-python\u002Findex",1785584462728]