[{"data":1,"prerenderedAt":3143},["ShallowReactive",2],{"doc:\u002Fadvanced-data-transformation-and-cleaning\u002Freading-excel-with-polars-and-arrow":3,"surround:\u002Fadvanced-data-transformation-and-cleaning\u002Freading-excel-with-polars-and-arrow":3134},{"id":4,"title":5,"body":6,"dateModified":3108,"datePublished":3109,"description":3110,"extension":3111,"faq":3112,"meta":3125,"navigation":259,"path":3126,"seo":3127,"slug":3130,"stem":3131,"type":3132,"__hash__":3133},"docs\u002Fadvanced-data-transformation-and-cleaning\u002Freading-excel-with-polars-and-arrow\u002Findex.md","Reading Excel with Polars and Arrow",{"type":7,"value":8,"toc":3089},"minimark",[9,24,180,185,188,229,232,300,307,313,406,409,471,486,490,493,715,718,722,725,756,763,852,856,859,1047,1064,1068,1071,1166,1173,1177,1180,1263,1266,1270,1273,1504,1523,1527,1530,1738,1754,1758,1761,2058,2139,2142,2146,2149,2292,2306,2310,2313,2617,2631,2635,2638,2718,2811,2817,2831,2835,2838,2927,2938,2942,2970,2974,2980,2986,2994,3008,3014,3018,3021,3025,3085],[10,11,12,13,17,18,23],"p",{},"pandas is not the only way to get a spreadsheet into Python any more. Polars — a DataFrame library built on Apache Arrow with a Rust core — reads Excel through the same fast ",[14,15,16],"code",{},"calamine"," engine that pandas can now use, then runs transformations across every core in a query engine that plans the work before it executes. For reporting jobs that read a workbook and immediately aggregate, join or reshape it, the combination is markedly quicker and considerably lighter on memory. This topic covers reading and writing Excel from Polars, the Arrow-backed formats that pair with it, and — just as important — the cases where pandas or openpyxl remain the right answer. It extends ",[19,20,22],"a",{"href":21},"\u002Fadvanced-data-transformation-and-cleaning\u002F","Advanced Data Transformation and Cleaning"," with the newer half of the ecosystem.",[25,26,34,35,34,39,34,43,34,50,34,57,34,65,34,71,34,76,34,84,34,89,34,94,34,98,34,102,34,106,34,109,34,113,34,118,34,121,34,127,34,131,34,134,34,137,34,145,34,151,34,156,34,159,34,162,34,166,34,171,34,176],"svg",{"viewBox":27,"role":28,"ariaLabelledBy":29,"xmlns":32,"style":33},"0 0 760 258","img",[30,31],"plx-t","plx-d","http:\u002F\u002Fwww.w3.org\u002F2000\u002Fsvg","width:100%;max-width:760px;height:auto;display:block;margin:1.5rem auto;font-family:Inter,ui-sans-serif,system-ui,sans-serif","\n  ",[36,37,38],"title",{"id":30},"How a workbook reaches a Polars DataFrame",[40,41,42],"desc",{"id":31},"The xlsx file is parsed by the Rust calamine reader into Arrow columns, which Polars queries with a multi-threaded engine before writing back to Excel or to Parquet.",[44,45],"rect",{"x":46,"y":46,"width":47,"height":48,"fill":49},"0","760","258","#ffffff",[51,52,56],"text",{"x":53,"y":54,"style":55},"380","26","font-size:13px;font-weight:600;fill:var(--muted,#5b6780);text-anchor:middle","One read, then columnar all the way",[44,58],{"x":54,"y":59,"width":60,"height":61,"rx":62,"fill":63,"stroke":64},"80","140","70","12","#f0f2f5","var(--line,#cdd5e6)",[51,66,70],{"x":67,"y":68,"style":69},"96","112","font-size:12.5px;font-weight:700;fill:var(--text,#172033);text-anchor:middle","sales.xlsx",[51,72,75],{"x":67,"y":73,"style":74},"132","font-size:11.5px;fill:var(--muted,#5b6780);text-anchor:middle","zip of XML",[77,78],"line",{"x1":79,"y1":80,"x2":81,"y2":80,"stroke":82,"style":83},"166","115","200","var(--brand,#5b5cf0)","stroke-width:2px",[85,86],"polygon",{"points":87,"fill":88},"200,115 190,110 190,120","#5b5cf0",[44,90],{"x":91,"y":59,"width":92,"height":61,"rx":62,"fill":93,"stroke":64},"204","150","#fdefd8",[51,95,16],{"x":96,"y":68,"style":97},"279","font-size:12.5px;font-weight:700;fill:var(--gold-ink,#7a4e06);text-anchor:middle",[51,99,101],{"x":96,"y":73,"style":100},"font-size:11.5px;fill:var(--gold-ink,#7a4e06);text-anchor:middle","Rust parser",[77,103],{"x1":104,"y1":80,"x2":105,"y2":80,"stroke":82,"style":83},"354","388",[85,107],{"points":108,"fill":88},"388,115 378,110 378,120",[44,110],{"x":111,"y":59,"width":92,"height":61,"rx":62,"fill":112,"stroke":64},"392","#ebebfd",[51,114,117],{"x":115,"y":68,"style":116},"467","font-size:12.5px;font-weight:700;fill:var(--brand-strong,#4338ca);text-anchor:middle","Arrow columns",[51,119,120],{"x":115,"y":73,"style":74},"typed, contiguous",[77,122],{"x1":123,"y1":124,"x2":125,"y2":61,"stroke":126,"style":83},"542","98","576","var(--teal,#0f9488)",[85,128],{"points":129,"fill":130},"576,70 569,81 564,73","#0f766e",[77,132],{"x1":123,"y1":73,"x2":125,"y2":133,"stroke":126,"style":83},"160",[85,135],{"points":136,"fill":130},"576,160 564,157 569,149",[44,138],{"x":139,"y":140,"width":141,"height":142,"rx":143,"fill":144,"stroke":64},"580","40","156","60","11","#d9f4f1",[51,146,150],{"x":147,"y":148,"style":149},"658","66","font-size:12.5px;font-weight:700;fill:var(--teal-ink,#0b6157);text-anchor:middle","write_excel",[51,152,155],{"x":147,"y":153,"style":154},"86","font-size:11.5px;fill:var(--teal-ink,#0b6157);text-anchor:middle","a styled report",[44,157],{"x":139,"y":158,"width":141,"height":142,"rx":143,"fill":144,"stroke":64},"130",[51,160,161],{"x":147,"y":141,"style":149},"write_parquet",[51,163,165],{"x":147,"y":164,"style":154},"176","a fast cache",[44,167],{"x":91,"y":164,"width":168,"height":169,"rx":170,"fill":63,"stroke":64},"338","52","10",[51,172,175],{"x":173,"y":81,"style":174},"373","font-size:12px;fill:var(--text,#172033);text-anchor:middle","filter, group_by, join — planned once,",[51,177,179],{"x":173,"y":178,"style":174},"218","then executed across every core",[181,182,184],"h2",{"id":183},"install-and-read-your-first-workbook","Install and read your first workbook",[10,186,187],{},"Polars reads Excel through an optional extra, so install the reader alongside it:",[189,190,195],"pre",{"className":191,"code":192,"language":193,"meta":194,"style":194},"language-bash shiki shiki-themes github-light github-dark-high-contrast","pip install \"polars[excel]\"        # polars + fastexcel\u002Fcalamine\npip install xlsxwriter             # only needed for write_excel\n","bash","",[14,196,197,216],{"__ignoreMap":194},[198,199,201,205,209,212],"span",{"class":77,"line":200},1,[198,202,204],{"class":203},"sMTad","pip",[198,206,208],{"class":207},"srMev"," install",[198,210,211],{"class":207}," \"polars[excel]\"",[198,213,215],{"class":214},"s-wDw","        # polars + fastexcel\u002Fcalamine\n",[198,217,219,221,223,226],{"class":77,"line":218},2,[198,220,204],{"class":203},[198,222,208],{"class":207},[198,224,225],{"class":207}," xlsxwriter",[198,227,228],{"class":214},"             # only needed for write_excel\n",[10,230,231],{},"The read itself mirrors pandas closely enough that most code translates on sight:",[189,233,237],{"className":234,"code":235,"language":236,"meta":194,"style":194},"language-python shiki shiki-themes github-light github-dark-high-contrast","import polars as pl\n\ndf = pl.read_excel(\"sales.xlsx\")                       # first sheet\nprint(df.head())\nprint(df.schema)\n","python",[14,238,239,255,261,282,292],{"__ignoreMap":194},[198,240,241,245,249,252],{"class":77,"line":200},[198,242,244],{"class":243},"s-kum","import",[198,246,248],{"class":247},"skGVy"," polars ",[198,250,251],{"class":243},"as",[198,253,254],{"class":247}," pl\n",[198,256,257],{"class":77,"line":218},[198,258,260],{"emptyLinePlaceholder":259},true,"\n",[198,262,264,267,270,273,276,279],{"class":77,"line":263},3,[198,265,266],{"class":247},"df ",[198,268,269],{"class":243},"=",[198,271,272],{"class":247}," pl.read_excel(",[198,274,275],{"class":207},"\"sales.xlsx\"",[198,277,278],{"class":247},")                       ",[198,280,281],{"class":214},"# first sheet\n",[198,283,285,289],{"class":77,"line":284},4,[198,286,288],{"class":287},"sP0c6","print",[198,290,291],{"class":247},"(df.head())\n",[198,293,295,297],{"class":77,"line":294},5,[198,296,288],{"class":287},[198,298,299],{"class":247},"(df.schema)\n",[189,301,305],{"className":302,"code":304,"language":51,"meta":194},[303],"language-text","shape: (5, 4)\n┌────────────┬────────┬─────────┬──────────┐\n│ order_date ┆ region ┆ product ┆ revenue  │\n│ ---        ┆ ---    ┆ ---     ┆ ---      │\n│ date       ┆ str    ┆ str     ┆ f64      │\n└────────────┴────────┴─────────┴──────────┘\n",[14,306,304],{"__ignoreMap":194},[10,308,309,312],{},[14,310,311],{},"schema"," is worth printing on every new file. Polars infers types once, at read time, and being explicit about what it found saves the class of bug where a column of order numbers silently becomes a float. Selecting sheets works as it does in pandas:",[189,314,316],{"className":234,"code":315,"language":236,"meta":194,"style":194},"q3 = pl.read_excel(\"sales.xlsx\", sheet_name=\"Q3\")\nby_position = pl.read_excel(\"sales.xlsx\", sheet_id=2)\neverything = pl.read_excel(\"sales.xlsx\", sheet_id=None)   # dict of DataFrames\nprint(list(everything))\n",[14,317,318,344,367,393],{"__ignoreMap":194},[198,319,320,323,325,327,329,332,336,338,341],{"class":77,"line":200},[198,321,322],{"class":247},"q3 ",[198,324,269],{"class":243},[198,326,272],{"class":247},[198,328,275],{"class":207},[198,330,331],{"class":247},", ",[198,333,335],{"class":334},"sa561","sheet_name",[198,337,269],{"class":243},[198,339,340],{"class":207},"\"Q3\"",[198,342,343],{"class":247},")\n",[198,345,346,349,351,353,355,357,360,362,365],{"class":77,"line":218},[198,347,348],{"class":247},"by_position ",[198,350,269],{"class":243},[198,352,272],{"class":247},[198,354,275],{"class":207},[198,356,331],{"class":247},[198,358,359],{"class":334},"sheet_id",[198,361,269],{"class":243},[198,363,364],{"class":287},"2",[198,366,343],{"class":247},[198,368,369,372,374,376,378,380,382,384,387,390],{"class":77,"line":263},[198,370,371],{"class":247},"everything ",[198,373,269],{"class":243},[198,375,272],{"class":247},[198,377,275],{"class":207},[198,379,331],{"class":247},[198,381,359],{"class":334},[198,383,269],{"class":243},[198,385,386],{"class":287},"None",[198,388,389],{"class":247},")   ",[198,391,392],{"class":214},"# dict of DataFrames\n",[198,394,395,397,400,403],{"class":77,"line":284},[198,396,288],{"class":287},[198,398,399],{"class":247},"(",[198,401,402],{"class":287},"list",[198,404,405],{"class":247},"(everything))\n",[10,407,408],{},"The reader's own options are passed through — for example to skip a banner row or to force a column to text:",[189,410,412],{"className":234,"code":411,"language":236,"meta":194,"style":194},"df = pl.read_excel(\n    \"sales.xlsx\",\n    read_options={\"header_row\": 2},\n    schema_overrides={\"order_id\": pl.String},\n)\n",[14,413,414,423,431,452,467],{"__ignoreMap":194},[198,415,416,418,420],{"class":77,"line":200},[198,417,266],{"class":247},[198,419,269],{"class":243},[198,421,422],{"class":247}," pl.read_excel(\n",[198,424,425,428],{"class":77,"line":218},[198,426,427],{"class":207},"    \"sales.xlsx\"",[198,429,430],{"class":247},",\n",[198,432,433,436,438,441,444,447,449],{"class":77,"line":263},[198,434,435],{"class":334},"    read_options",[198,437,269],{"class":243},[198,439,440],{"class":247},"{",[198,442,443],{"class":207},"\"header_row\"",[198,445,446],{"class":247},": ",[198,448,364],{"class":287},[198,450,451],{"class":247},"},\n",[198,453,454,457,459,461,464],{"class":77,"line":284},[198,455,456],{"class":334},"    schema_overrides",[198,458,269],{"class":243},[198,460,440],{"class":247},[198,462,463],{"class":207},"\"order_id\"",[198,465,466],{"class":247},": pl.String},\n",[198,468,469],{"class":77,"line":294},[198,470,343],{"class":247},[10,472,473,476,477,480,481,485],{},[14,474,475],{},"schema_overrides"," is the Polars equivalent of pandas' ",[14,478,479],{},"dtype=",", and it is the right tool for the perennial problem of leading zeros in reference codes — the same issue tackled from the pandas side in ",[19,482,484],{"href":483},"\u002Fadvanced-data-transformation-and-cleaning\u002Fcleaning-excel-data-with-pandas\u002Fconvert-excel-text-columns-to-numbers-with-pandas\u002F","Convert Excel text columns to numbers with pandas",".",[181,487,489],{"id":488},"transform-with-expressions-instead-of-loops","Transform with expressions instead of loops",[10,491,492],{},"The reason to read into Polars is what comes next. Polars expressions describe a transformation, and the engine plans and parallelises it rather than executing statement by statement:",[189,494,496],{"className":234,"code":495,"language":236,"meta":194,"style":194},"import polars as pl\n\ndf = pl.read_excel(\"sales.xlsx\")\n\nsummary = (\n    df.filter(pl.col(\"revenue\") > 0)\n      .with_columns(\n          month=pl.col(\"order_date\").dt.strftime(\"%Y-%m\"),\n          net=pl.col(\"revenue\") * 0.8,\n      )\n      .group_by(\"region\", \"month\")\n      .agg(\n          orders=pl.len(),\n          revenue=pl.col(\"revenue\").sum().round(2),\n          best=pl.col(\"product\").mode().first(),\n      )\n      .sort(\"region\", \"month\")\n)\nprint(summary)\n",[14,497,498,508,512,524,528,538,558,564,587,609,615,631,637,648,667,683,688,702,707],{"__ignoreMap":194},[198,499,500,502,504,506],{"class":77,"line":200},[198,501,244],{"class":243},[198,503,248],{"class":247},[198,505,251],{"class":243},[198,507,254],{"class":247},[198,509,510],{"class":77,"line":218},[198,511,260],{"emptyLinePlaceholder":259},[198,513,514,516,518,520,522],{"class":77,"line":263},[198,515,266],{"class":247},[198,517,269],{"class":243},[198,519,272],{"class":247},[198,521,275],{"class":207},[198,523,343],{"class":247},[198,525,526],{"class":77,"line":284},[198,527,260],{"emptyLinePlaceholder":259},[198,529,530,533,535],{"class":77,"line":294},[198,531,532],{"class":247},"summary ",[198,534,269],{"class":243},[198,536,537],{"class":247}," (\n",[198,539,541,544,547,550,553,556],{"class":77,"line":540},6,[198,542,543],{"class":247},"    df.filter(pl.col(",[198,545,546],{"class":207},"\"revenue\"",[198,548,549],{"class":247},") ",[198,551,552],{"class":243},">",[198,554,555],{"class":287}," 0",[198,557,343],{"class":247},[198,559,561],{"class":77,"line":560},7,[198,562,563],{"class":247},"      .with_columns(\n",[198,565,567,570,572,575,578,581,584],{"class":77,"line":566},8,[198,568,569],{"class":334},"          month",[198,571,269],{"class":243},[198,573,574],{"class":247},"pl.col(",[198,576,577],{"class":207},"\"order_date\"",[198,579,580],{"class":247},").dt.strftime(",[198,582,583],{"class":207},"\"%Y-%m\"",[198,585,586],{"class":247},"),\n",[198,588,590,593,595,597,599,601,604,607],{"class":77,"line":589},9,[198,591,592],{"class":334},"          net",[198,594,269],{"class":243},[198,596,574],{"class":247},[198,598,546],{"class":207},[198,600,549],{"class":247},[198,602,603],{"class":243},"*",[198,605,606],{"class":287}," 0.8",[198,608,430],{"class":247},[198,610,612],{"class":77,"line":611},10,[198,613,614],{"class":247},"      )\n",[198,616,618,621,624,626,629],{"class":77,"line":617},11,[198,619,620],{"class":247},"      .group_by(",[198,622,623],{"class":207},"\"region\"",[198,625,331],{"class":247},[198,627,628],{"class":207},"\"month\"",[198,630,343],{"class":247},[198,632,634],{"class":77,"line":633},12,[198,635,636],{"class":247},"      .agg(\n",[198,638,640,643,645],{"class":77,"line":639},13,[198,641,642],{"class":334},"          orders",[198,644,269],{"class":243},[198,646,647],{"class":247},"pl.len(),\n",[198,649,651,654,656,658,660,663,665],{"class":77,"line":650},14,[198,652,653],{"class":334},"          revenue",[198,655,269],{"class":243},[198,657,574],{"class":247},[198,659,546],{"class":207},[198,661,662],{"class":247},").sum().round(",[198,664,364],{"class":287},[198,666,586],{"class":247},[198,668,670,673,675,677,680],{"class":77,"line":669},15,[198,671,672],{"class":334},"          best",[198,674,269],{"class":243},[198,676,574],{"class":247},[198,678,679],{"class":207},"\"product\"",[198,681,682],{"class":247},").mode().first(),\n",[198,684,686],{"class":77,"line":685},16,[198,687,614],{"class":247},[198,689,691,694,696,698,700],{"class":77,"line":690},17,[198,692,693],{"class":247},"      .sort(",[198,695,623],{"class":207},[198,697,331],{"class":247},[198,699,628],{"class":207},[198,701,343],{"class":247},[198,703,705],{"class":77,"line":704},18,[198,706,343],{"class":247},[198,708,710,712],{"class":77,"line":709},19,[198,711,288],{"class":287},[198,713,714],{"class":247},"(summary)\n",[10,716,717],{},"Every step above is a column operation, so nothing iterates in Python. The equivalent pandas code is comparable in length but executes single-threaded per operation, and materialises an intermediate frame at each step. For a workbook of a few thousand rows the difference is imperceptible; over a few million rows joined against a database extract, it is the difference between a report that runs in a coffee break and one that does not.",[181,719,721],{"id":720},"convert-between-polars-and-pandas-without-paying-for-it","Convert between Polars and pandas without paying for it",[10,723,724],{},"Adoption does not have to be all-or-nothing. Because both libraries speak Arrow, converting is close to free — no serialisation, and often no copy at all:",[189,726,728],{"className":234,"code":727,"language":236,"meta":194,"style":194},"pdf = summary.to_pandas()          # hand off to existing formatting code\nback = pl.from_pandas(pdf)         # and back again\n",[14,729,730,743],{"__ignoreMap":194},[198,731,732,735,737,740],{"class":77,"line":200},[198,733,734],{"class":247},"pdf ",[198,736,269],{"class":243},[198,738,739],{"class":247}," summary.to_pandas()          ",[198,741,742],{"class":214},"# hand off to existing formatting code\n",[198,744,745,748,750,753],{"class":77,"line":218},[198,746,747],{"class":247},"back ",[198,749,269],{"class":243},[198,751,752],{"class":247}," pl.from_pandas(pdf)         ",[198,754,755],{"class":214},"# and back again\n",[10,757,758,759,485],{},"That makes an incremental path realistic: read and reshape in Polars where the data is large, convert to pandas for the last mile, and keep every line of existing openpyxl or xlsxwriter styling code untouched. The formatting layer is covered across ",[19,760,762],{"href":761},"\u002Fformatting-and-charting-excel-reports-with-python\u002F","Formatting and Charting Excel Reports with Python",[25,764,34,769,34,772,34,775,34,778,34,781,34,786,34,792,34,797,34,800,34,804,34,808,34,811,34,816,34,819,34,822,34,825,34,828,34,832,34,837,34,840,34,843,34,846,34,849],{"viewBox":765,"role":28,"ariaLabelledBy":766,"xmlns":32,"style":33},"0 0 760 246",[767,768],"plx2-t","plx2-d",[36,770,771],{"id":767},"Which library owns which stage of a reporting job",[40,773,774],{"id":768},"Polars is strongest at reading and transforming large extracts, pandas remains convenient for small frames and its ecosystem, and openpyxl or xlsxwriter own the styled output.",[44,776],{"x":46,"y":46,"width":47,"height":777,"fill":49},"246",[51,779,780],{"x":53,"y":54,"style":55},"A pipeline can use all three without conflict",[44,782],{"x":783,"y":784,"width":785,"height":92,"rx":62,"fill":112,"stroke":64},"28","46","226",[51,787,791],{"x":788,"y":789,"style":790},"141","76","font-size:13px;font-weight:700;fill:var(--brand-strong,#4338ca);text-anchor:middle","Polars",[51,793,796],{"x":788,"y":794,"style":795},"106","font-size:11.5px;fill:var(--text,#172033);text-anchor:middle","read large workbooks",[51,798,799],{"x":788,"y":158,"style":795},"join, group, reshape",[51,801,803],{"x":788,"y":802,"style":795},"154","lazy scans over folders",[51,805,807],{"x":788,"y":806,"style":74},"178","multi-threaded by default",[44,809],{"x":810,"y":784,"width":785,"height":92,"rx":62,"fill":144,"stroke":64},"266",[51,812,815],{"x":813,"y":789,"style":814},"379","font-size:13px;font-weight:700;fill:var(--teal-ink,#0b6157);text-anchor:middle","pandas",[51,817,818],{"x":813,"y":794,"style":795},"small and medium frames",[51,820,821],{"x":813,"y":158,"style":795},"pivot_table, plotting",[51,823,824],{"x":813,"y":802,"style":795},"the widest ecosystem",[51,826,827],{"x":813,"y":806,"style":74},"what your team already knows",[44,829],{"x":830,"y":784,"width":831,"height":92,"rx":62,"fill":93,"stroke":64},"504","228",[51,833,836],{"x":834,"y":789,"style":835},"618","font-size:13px;font-weight:700;fill:var(--gold-ink,#7a4e06);text-anchor:middle","openpyxl \u002F xlsxwriter",[51,838,839],{"x":834,"y":794,"style":795},"styles and number formats",[51,841,842],{"x":834,"y":158,"style":795},"charts, tables, images",[51,844,845],{"x":834,"y":802,"style":795},"editing existing files",[51,847,848],{"x":834,"y":806,"style":74},"the only route to a template",[51,850,851],{"x":53,"y":831,"style":74},"Arrow makes the hand-offs between the first two effectively free",[181,853,855],{"id":854},"scan-a-folder-lazily-instead-of-reading-it-all","Scan a folder lazily instead of reading it all",[10,857,858],{},"Polars' lazy API plans a whole pipeline before touching data, which lets it push filters down and read only the columns a query needs. Excel itself cannot be scanned lazily — the format has to be parsed in full — but the pattern still pays once workbooks are converted to Parquet:",[189,860,862],{"className":234,"code":861,"language":236,"meta":194,"style":194},"from pathlib import Path\n\nimport polars as pl\n\n# One-off: convert a folder of monthly workbooks to Parquet.\nfor path in sorted(Path(\"monthly\").glob(\"*.xlsx\")):\n    pl.read_excel(path).write_parquet(path.with_suffix(\".parquet\"))\n\n# Every run after that: scan lazily, filter early, collect once.\nreport = (\n    pl.scan_parquet(\"monthly\u002F*.parquet\")\n      .filter(pl.col(\"region\") == \"EMEA\")\n      .group_by(\"product\")\n      .agg(revenue=pl.col(\"revenue\").sum())\n      .sort(\"revenue\", descending=True)\n      .head(20)\n      .collect()\n)\n",[14,863,864,877,881,891,895,900,929,940,944,949,958,968,985,993,1010,1028,1038,1043],{"__ignoreMap":194},[198,865,866,869,872,874],{"class":77,"line":200},[198,867,868],{"class":243},"from",[198,870,871],{"class":247}," pathlib ",[198,873,244],{"class":243},[198,875,876],{"class":247}," Path\n",[198,878,879],{"class":77,"line":218},[198,880,260],{"emptyLinePlaceholder":259},[198,882,883,885,887,889],{"class":77,"line":263},[198,884,244],{"class":243},[198,886,248],{"class":247},[198,888,251],{"class":243},[198,890,254],{"class":247},[198,892,893],{"class":77,"line":284},[198,894,260],{"emptyLinePlaceholder":259},[198,896,897],{"class":77,"line":294},[198,898,899],{"class":214},"# One-off: convert a folder of monthly workbooks to Parquet.\n",[198,901,902,905,908,911,914,917,920,923,926],{"class":77,"line":540},[198,903,904],{"class":243},"for",[198,906,907],{"class":247}," path ",[198,909,910],{"class":243},"in",[198,912,913],{"class":287}," sorted",[198,915,916],{"class":247},"(Path(",[198,918,919],{"class":207},"\"monthly\"",[198,921,922],{"class":247},").glob(",[198,924,925],{"class":207},"\"*.xlsx\"",[198,927,928],{"class":247},")):\n",[198,930,931,934,937],{"class":77,"line":560},[198,932,933],{"class":247},"    pl.read_excel(path).write_parquet(path.with_suffix(",[198,935,936],{"class":207},"\".parquet\"",[198,938,939],{"class":247},"))\n",[198,941,942],{"class":77,"line":566},[198,943,260],{"emptyLinePlaceholder":259},[198,945,946],{"class":77,"line":589},[198,947,948],{"class":214},"# Every run after that: scan lazily, filter early, collect once.\n",[198,950,951,954,956],{"class":77,"line":611},[198,952,953],{"class":247},"report ",[198,955,269],{"class":243},[198,957,537],{"class":247},[198,959,960,963,966],{"class":77,"line":617},[198,961,962],{"class":247},"    pl.scan_parquet(",[198,964,965],{"class":207},"\"monthly\u002F*.parquet\"",[198,967,343],{"class":247},[198,969,970,973,975,977,980,983],{"class":77,"line":633},[198,971,972],{"class":247},"      .filter(pl.col(",[198,974,623],{"class":207},[198,976,549],{"class":247},[198,978,979],{"class":243},"==",[198,981,982],{"class":207}," \"EMEA\"",[198,984,343],{"class":247},[198,986,987,989,991],{"class":77,"line":639},[198,988,620],{"class":247},[198,990,679],{"class":207},[198,992,343],{"class":247},[198,994,995,998,1001,1003,1005,1007],{"class":77,"line":650},[198,996,997],{"class":247},"      .agg(",[198,999,1000],{"class":334},"revenue",[198,1002,269],{"class":243},[198,1004,574],{"class":247},[198,1006,546],{"class":207},[198,1008,1009],{"class":247},").sum())\n",[198,1011,1012,1014,1016,1018,1021,1023,1026],{"class":77,"line":669},[198,1013,693],{"class":247},[198,1015,546],{"class":207},[198,1017,331],{"class":247},[198,1019,1020],{"class":334},"descending",[198,1022,269],{"class":243},[198,1024,1025],{"class":287},"True",[198,1027,343],{"class":247},[198,1029,1030,1033,1036],{"class":77,"line":685},[198,1031,1032],{"class":247},"      .head(",[198,1034,1035],{"class":287},"20",[198,1037,343],{"class":247},[198,1039,1040],{"class":77,"line":690},[198,1041,1042],{"class":247},"      .collect()\n",[198,1044,1045],{"class":77,"line":704},[198,1046,343],{"class":247},[10,1048,1049,1050,331,1053,1056,1057,1059,1060,485],{},"The lazy scan reads only the ",[14,1051,1052],{},"region",[14,1054,1055],{},"product"," and ",[14,1058,1000],{}," columns, and only the row groups that can contain EMEA rows. That is the single biggest speed-up available to a job that repeatedly re-reads the same historical workbooks — covered step by step in ",[19,1061,1063],{"href":1062},"\u002Fadvanced-data-transformation-and-cleaning\u002Freading-excel-with-polars-and-arrow\u002Fconvert-excel-files-to-parquet-with-python\u002F","Convert Excel files to Parquet with Python",[181,1065,1067],{"id":1066},"write-excel-back-out","Write Excel back out",[10,1069,1070],{},"Polars can write a workbook directly, driving xlsxwriter beneath the API, which is enough for most delivered reports:",[189,1072,1074],{"className":234,"code":1073,"language":236,"meta":194,"style":194},"summary.write_excel(\n    \"regional_summary.xlsx\",\n    worksheet=\"Summary\",\n    table_style=\"Table Style Medium 9\",\n    autofit=True,\n    column_formats={\"revenue\": \"#,##0.00\", \"net\": \"#,##0.00\"},\n    freeze_panes=\"A2\",\n)\n",[14,1075,1076,1081,1088,1100,1112,1123,1150,1162],{"__ignoreMap":194},[198,1077,1078],{"class":77,"line":200},[198,1079,1080],{"class":247},"summary.write_excel(\n",[198,1082,1083,1086],{"class":77,"line":218},[198,1084,1085],{"class":207},"    \"regional_summary.xlsx\"",[198,1087,430],{"class":247},[198,1089,1090,1093,1095,1098],{"class":77,"line":263},[198,1091,1092],{"class":334},"    worksheet",[198,1094,269],{"class":243},[198,1096,1097],{"class":207},"\"Summary\"",[198,1099,430],{"class":247},[198,1101,1102,1105,1107,1110],{"class":77,"line":284},[198,1103,1104],{"class":334},"    table_style",[198,1106,269],{"class":243},[198,1108,1109],{"class":207},"\"Table Style Medium 9\"",[198,1111,430],{"class":247},[198,1113,1114,1117,1119,1121],{"class":77,"line":294},[198,1115,1116],{"class":334},"    autofit",[198,1118,269],{"class":243},[198,1120,1025],{"class":287},[198,1122,430],{"class":247},[198,1124,1125,1128,1130,1132,1134,1136,1139,1141,1144,1146,1148],{"class":77,"line":540},[198,1126,1127],{"class":334},"    column_formats",[198,1129,269],{"class":243},[198,1131,440],{"class":247},[198,1133,546],{"class":207},[198,1135,446],{"class":247},[198,1137,1138],{"class":207},"\"#,##0.00\"",[198,1140,331],{"class":247},[198,1142,1143],{"class":207},"\"net\"",[198,1145,446],{"class":247},[198,1147,1138],{"class":207},[198,1149,451],{"class":247},[198,1151,1152,1155,1157,1160],{"class":77,"line":560},[198,1153,1154],{"class":334},"    freeze_panes",[198,1156,269],{"class":243},[198,1158,1159],{"class":207},"\"A2\"",[198,1161,430],{"class":247},[198,1163,1164],{"class":77,"line":566},[198,1165,343],{"class":247},[10,1167,1168,1169,485],{},"That single call produces a formatted Excel table with number formats, autofitted columns and a frozen header — the same output that takes a couple of dozen lines through openpyxl. What it cannot do is edit an existing workbook or fill a template, because xlsxwriter only creates new files. For those, keep openpyxl: ",[19,1170,1172],{"href":1171},"\u002Fautomating-reporting-workflows\u002Fgenerating-excel-reports-from-templates\u002Fpopulate-excel-template-without-losing-formatting\u002F","Populate an Excel template without losing formatting",[181,1174,1176],{"id":1175},"know-the-limits-before-you-commit","Know the limits before you commit",[10,1178,1179],{},"Polars is a data-processing library, not a spreadsheet library. It reads values, not presentation, and that shapes where it fits:",[1181,1182,1183,1198],"table",{},[1184,1185,1186],"thead",{},[1187,1188,1189,1193,1195],"tr",{},[1190,1191,1192],"th",{},"Task",[1190,1194,791],{},[1190,1196,1197],{},"Use instead",[1199,1200,1201,1213,1224,1233,1244,1254],"tbody",{},[1187,1202,1203,1207,1210],{},[1204,1205,1206],"td",{},"Read a large sheet fast",[1204,1208,1209],{},"Yes",[1204,1211,1212],{},"—",[1187,1214,1215,1218,1221],{},[1204,1216,1217],{},"Read cell colours, comments, merged regions",[1204,1219,1220],{},"No",[1204,1222,1223],{},"openpyxl",[1187,1225,1226,1229,1231],{},[1204,1227,1228],{},"Edit one cell in an existing workbook",[1204,1230,1220],{},[1204,1232,1223],{},[1187,1234,1235,1238,1241],{},[1204,1236,1237],{},"Write a formatted new report",[1204,1239,1240],{},"Yes, via xlsxwriter",[1204,1242,1243],{},"openpyxl for templates",[1187,1245,1246,1249,1251],{},[1204,1247,1248],{},"Add a native pivot table or chart",[1204,1250,1220],{},[1204,1252,1253],{},"openpyxl or xlsxwriter",[1187,1255,1256,1259,1261],{},[1204,1257,1258],{},"Group, join, aggregate millions of rows",[1204,1260,1209],{},[1204,1262,1212],{},[10,1264,1265],{},"The honest summary is that Polars replaces the pandas half of a reporting job, not the openpyxl half. A typical modern pipeline reads with Polars, aggregates with Polars, and hands the result to xlsxwriter or openpyxl for presentation.",[181,1267,1269],{"id":1268},"clean-the-messy-parts-of-a-real-sheet","Clean the messy parts of a real sheet",[10,1271,1272],{},"Workbooks that come from people rather than systems need the same cleaning in Polars as in pandas — banner rows, trailing totals, text in numeric columns, inconsistent casing. The idioms differ, so here are the ones that come up in nearly every report:",[189,1274,1276],{"className":234,"code":1275,"language":236,"meta":194,"style":194},"import polars as pl\n\nraw = pl.read_excel(\"messy.xlsx\", read_options={\"header_row\": 3})\n\nclean = (\n    raw\n    # drop rows that are entirely empty\n    .filter(~pl.all_horizontal(pl.all().is_null()))\n    # drop a trailing \"Total\" row that the exporter appended\n    .filter(pl.col(\"region\") != \"Total\")\n    # normalise text columns in one pass\n    .with_columns(\n        pl.col(pl.String).str.strip_chars().str.to_titlecase(),\n    )\n    # coerce a column that arrived as \"1,234.50\" text\n    .with_columns(\n        pl.col(\"revenue\").cast(pl.String)\n          .str.replace_all(\",\", \"\")\n          .cast(pl.Float64, strict=False)\n          .alias(\"revenue\"),\n    )\n    # give every column a snake_case name\n    .rename(lambda name: name.strip().lower().replace(\" \", \"_\"))\n)\nprint(clean.null_count())\n",[14,1277,1278,1288,1292,1323,1327,1336,1341,1346,1357,1362,1379,1384,1389,1394,1399,1404,1408,1418,1433,1448,1458,1463,1469,1491,1496],{"__ignoreMap":194},[198,1279,1280,1282,1284,1286],{"class":77,"line":200},[198,1281,244],{"class":243},[198,1283,248],{"class":247},[198,1285,251],{"class":243},[198,1287,254],{"class":247},[198,1289,1290],{"class":77,"line":218},[198,1291,260],{"emptyLinePlaceholder":259},[198,1293,1294,1297,1299,1301,1304,1306,1309,1311,1313,1315,1317,1320],{"class":77,"line":263},[198,1295,1296],{"class":247},"raw ",[198,1298,269],{"class":243},[198,1300,272],{"class":247},[198,1302,1303],{"class":207},"\"messy.xlsx\"",[198,1305,331],{"class":247},[198,1307,1308],{"class":334},"read_options",[198,1310,269],{"class":243},[198,1312,440],{"class":247},[198,1314,443],{"class":207},[198,1316,446],{"class":247},[198,1318,1319],{"class":287},"3",[198,1321,1322],{"class":247},"})\n",[198,1324,1325],{"class":77,"line":284},[198,1326,260],{"emptyLinePlaceholder":259},[198,1328,1329,1332,1334],{"class":77,"line":294},[198,1330,1331],{"class":247},"clean ",[198,1333,269],{"class":243},[198,1335,537],{"class":247},[198,1337,1338],{"class":77,"line":540},[198,1339,1340],{"class":247},"    raw\n",[198,1342,1343],{"class":77,"line":560},[198,1344,1345],{"class":214},"    # drop rows that are entirely empty\n",[198,1347,1348,1351,1354],{"class":77,"line":566},[198,1349,1350],{"class":247},"    .filter(",[198,1352,1353],{"class":243},"~",[198,1355,1356],{"class":247},"pl.all_horizontal(pl.all().is_null()))\n",[198,1358,1359],{"class":77,"line":589},[198,1360,1361],{"class":214},"    # drop a trailing \"Total\" row that the exporter appended\n",[198,1363,1364,1367,1369,1371,1374,1377],{"class":77,"line":611},[198,1365,1366],{"class":247},"    .filter(pl.col(",[198,1368,623],{"class":207},[198,1370,549],{"class":247},[198,1372,1373],{"class":243},"!=",[198,1375,1376],{"class":207}," \"Total\"",[198,1378,343],{"class":247},[198,1380,1381],{"class":77,"line":617},[198,1382,1383],{"class":214},"    # normalise text columns in one pass\n",[198,1385,1386],{"class":77,"line":633},[198,1387,1388],{"class":247},"    .with_columns(\n",[198,1390,1391],{"class":77,"line":639},[198,1392,1393],{"class":247},"        pl.col(pl.String).str.strip_chars().str.to_titlecase(),\n",[198,1395,1396],{"class":77,"line":650},[198,1397,1398],{"class":247},"    )\n",[198,1400,1401],{"class":77,"line":669},[198,1402,1403],{"class":214},"    # coerce a column that arrived as \"1,234.50\" text\n",[198,1405,1406],{"class":77,"line":685},[198,1407,1388],{"class":247},[198,1409,1410,1413,1415],{"class":77,"line":690},[198,1411,1412],{"class":247},"        pl.col(",[198,1414,546],{"class":207},[198,1416,1417],{"class":247},").cast(pl.String)\n",[198,1419,1420,1423,1426,1428,1431],{"class":77,"line":704},[198,1421,1422],{"class":247},"          .str.replace_all(",[198,1424,1425],{"class":207},"\",\"",[198,1427,331],{"class":247},[198,1429,1430],{"class":207},"\"\"",[198,1432,343],{"class":247},[198,1434,1435,1438,1441,1443,1446],{"class":77,"line":709},[198,1436,1437],{"class":247},"          .cast(pl.Float64, ",[198,1439,1440],{"class":334},"strict",[198,1442,269],{"class":243},[198,1444,1445],{"class":287},"False",[198,1447,343],{"class":247},[198,1449,1451,1454,1456],{"class":77,"line":1450},20,[198,1452,1453],{"class":247},"          .alias(",[198,1455,546],{"class":207},[198,1457,586],{"class":247},[198,1459,1461],{"class":77,"line":1460},21,[198,1462,1398],{"class":247},[198,1464,1466],{"class":77,"line":1465},22,[198,1467,1468],{"class":214},"    # give every column a snake_case name\n",[198,1470,1472,1475,1478,1481,1484,1486,1489],{"class":77,"line":1471},23,[198,1473,1474],{"class":247},"    .rename(",[198,1476,1477],{"class":243},"lambda",[198,1479,1480],{"class":247}," name: name.strip().lower().replace(",[198,1482,1483],{"class":207},"\" \"",[198,1485,331],{"class":247},[198,1487,1488],{"class":207},"\"_\"",[198,1490,939],{"class":247},[198,1492,1494],{"class":77,"line":1493},24,[198,1495,343],{"class":247},[198,1497,1499,1501],{"class":77,"line":1498},25,[198,1500,288],{"class":287},[198,1502,1503],{"class":247},"(clean.null_count())\n",[10,1505,1506,1507,1510,1511,1514,1515,1518,1519,485],{},"Two Polars-specific conveniences are worth noting. ",[14,1508,1509],{},"pl.col(pl.String)"," selects every column of a given type, so one expression normalises all text columns without naming them. And ",[14,1512,1513],{},"strict=False"," on a cast turns unparseable values into nulls rather than raising, which is the behaviour you want when a single stray footnote would otherwise abort the whole read. ",[14,1516,1517],{},"null_count()"," afterwards tells you how many values that cost — the Polars counterpart to the audit in ",[19,1520,1522],{"href":1521},"\u002Fadvanced-data-transformation-and-cleaning\u002Fhandling-missing-data-in-excel-reports\u002Ffind-and-report-missing-values-in-an-excel-file\u002F","Find and report missing values in an Excel file",[181,1524,1526],{"id":1525},"join-a-workbook-against-a-database-extract","Join a workbook against a database extract",[10,1528,1529],{},"Reporting rarely uses one source. A common shape is a spreadsheet of manual adjustments joined onto a query result, and Polars joins are both fast and strict about what they do with unmatched rows:",[189,1531,1533],{"className":234,"code":1532,"language":236,"meta":194,"style":194},"import polars as pl\n\nadjustments = pl.read_excel(\"adjustments.xlsx\")\nfacts = pl.read_database_uri(\n    query=\"SELECT order_id, region, revenue FROM orders WHERE order_date >= '2026-01-01'\",\n    uri=\"postgresql:\u002F\u002Freporting:secret@db.internal\u002Fanalytics\",\n)\n\nmerged = facts.join(adjustments, on=\"order_id\", how=\"left\")\nunmatched = adjustments.join(facts, on=\"order_id\", how=\"anti\")\n\nprint(f\"{len(unmatched)} adjustment rows matched nothing\")\nfinal = merged.with_columns(\n    revenue=pl.col(\"revenue\") + pl.col(\"adjustment\").fill_null(0.0)\n)\n",[14,1534,1535,1545,1549,1563,1573,1585,1597,1601,1605,1634,1661,1665,1694,1704,1734],{"__ignoreMap":194},[198,1536,1537,1539,1541,1543],{"class":77,"line":200},[198,1538,244],{"class":243},[198,1540,248],{"class":247},[198,1542,251],{"class":243},[198,1544,254],{"class":247},[198,1546,1547],{"class":77,"line":218},[198,1548,260],{"emptyLinePlaceholder":259},[198,1550,1551,1554,1556,1558,1561],{"class":77,"line":263},[198,1552,1553],{"class":247},"adjustments ",[198,1555,269],{"class":243},[198,1557,272],{"class":247},[198,1559,1560],{"class":207},"\"adjustments.xlsx\"",[198,1562,343],{"class":247},[198,1564,1565,1568,1570],{"class":77,"line":284},[198,1566,1567],{"class":247},"facts ",[198,1569,269],{"class":243},[198,1571,1572],{"class":247}," pl.read_database_uri(\n",[198,1574,1575,1578,1580,1583],{"class":77,"line":294},[198,1576,1577],{"class":334},"    query",[198,1579,269],{"class":243},[198,1581,1582],{"class":207},"\"SELECT order_id, region, revenue FROM orders WHERE order_date >= '2026-01-01'\"",[198,1584,430],{"class":247},[198,1586,1587,1590,1592,1595],{"class":77,"line":540},[198,1588,1589],{"class":334},"    uri",[198,1591,269],{"class":243},[198,1593,1594],{"class":207},"\"postgresql:\u002F\u002Freporting:secret@db.internal\u002Fanalytics\"",[198,1596,430],{"class":247},[198,1598,1599],{"class":77,"line":560},[198,1600,343],{"class":247},[198,1602,1603],{"class":77,"line":566},[198,1604,260],{"emptyLinePlaceholder":259},[198,1606,1607,1610,1612,1615,1618,1620,1622,1624,1627,1629,1632],{"class":77,"line":589},[198,1608,1609],{"class":247},"merged ",[198,1611,269],{"class":243},[198,1613,1614],{"class":247}," facts.join(adjustments, ",[198,1616,1617],{"class":334},"on",[198,1619,269],{"class":243},[198,1621,463],{"class":207},[198,1623,331],{"class":247},[198,1625,1626],{"class":334},"how",[198,1628,269],{"class":243},[198,1630,1631],{"class":207},"\"left\"",[198,1633,343],{"class":247},[198,1635,1636,1639,1641,1644,1646,1648,1650,1652,1654,1656,1659],{"class":77,"line":611},[198,1637,1638],{"class":247},"unmatched ",[198,1640,269],{"class":243},[198,1642,1643],{"class":247}," adjustments.join(facts, ",[198,1645,1617],{"class":334},[198,1647,269],{"class":243},[198,1649,463],{"class":207},[198,1651,331],{"class":247},[198,1653,1626],{"class":334},[198,1655,269],{"class":243},[198,1657,1658],{"class":207},"\"anti\"",[198,1660,343],{"class":247},[198,1662,1663],{"class":77,"line":617},[198,1664,260],{"emptyLinePlaceholder":259},[198,1666,1667,1669,1671,1674,1677,1680,1683,1686,1689,1692],{"class":77,"line":633},[198,1668,288],{"class":287},[198,1670,399],{"class":247},[198,1672,1673],{"class":243},"f",[198,1675,1676],{"class":207},"\"",[198,1678,440],{"class":1679},"sSjpA",[198,1681,1682],{"class":287},"len",[198,1684,1685],{"class":247},"(unmatched)",[198,1687,1688],{"class":1679},"}",[198,1690,1691],{"class":207}," adjustment rows matched nothing\"",[198,1693,343],{"class":247},[198,1695,1696,1699,1701],{"class":77,"line":639},[198,1697,1698],{"class":247},"final ",[198,1700,269],{"class":243},[198,1702,1703],{"class":247}," merged.with_columns(\n",[198,1705,1706,1709,1711,1713,1715,1717,1720,1723,1726,1729,1732],{"class":77,"line":650},[198,1707,1708],{"class":334},"    revenue",[198,1710,269],{"class":243},[198,1712,574],{"class":247},[198,1714,546],{"class":207},[198,1716,549],{"class":247},[198,1718,1719],{"class":243},"+",[198,1721,1722],{"class":247}," pl.col(",[198,1724,1725],{"class":207},"\"adjustment\"",[198,1727,1728],{"class":247},").fill_null(",[198,1730,1731],{"class":287},"0.0",[198,1733,343],{"class":247},[198,1735,1736],{"class":77,"line":669},[198,1737,343],{"class":247},[10,1739,1740,1741,1744,1745,1749,1750,485],{},"The ",[14,1742,1743],{},"anti"," join is the piece worth stealing regardless of library: it names the rows from the spreadsheet that found no partner, which is exactly the reconciliation question a finance reviewer will ask. The pandas equivalent is in ",[19,1746,1748],{"href":1747},"\u002Fadvanced-data-transformation-and-cleaning\u002Fmerging-and-joining-excel-dataframes\u002Ffind-rows-in-one-excel-file-missing-from-another\u002F","Find rows in one Excel file missing from another",", and the database side in ",[19,1751,1753],{"href":1752},"\u002Fadvanced-data-transformation-and-cleaning\u002Fmoving-data-between-excel-and-databases\u002Fexport-sql-query-results-to-excel-with-python\u002F","Export SQL query results to Excel with Python",[181,1755,1757],{"id":1756},"measure-it-on-your-own-data-before-switching","Measure it on your own data before switching",[10,1759,1760],{},"Published benchmarks are run on other people's files. The honest way to decide is a five-line measurement on the workbook your job actually reads, comparing the full path — read plus transformation — rather than the read alone:",[189,1762,1764],{"className":234,"code":1763,"language":236,"meta":194,"style":194},"\"\"\"Compare a real pipeline in both libraries.\"\"\"\nimport time\n\nimport pandas as pd\nimport polars as pl\n\ndef timed(label, fn):\n    start = time.perf_counter()\n    result = fn()\n    print(f\"{label:24s} {time.perf_counter() - start:6.2f}s  rows={len(result)}\")\n    return result\n\ntimed(\"pandas openpyxl\", lambda: (\n    pd.read_excel(\"sales.xlsx\", engine=\"openpyxl\")\n      .groupby(\"region\", as_index=False)[\"revenue\"].sum()\n))\ntimed(\"pandas calamine\", lambda: (\n    pd.read_excel(\"sales.xlsx\", engine=\"calamine\")\n      .groupby(\"region\", as_index=False)[\"revenue\"].sum()\n))\ntimed(\"polars\", lambda: (\n    pl.read_excel(\"sales.xlsx\").group_by(\"region\").agg(pl.col(\"revenue\").sum())\n))\n",[14,1765,1766,1771,1778,1782,1794,1804,1808,1820,1830,1840,1894,1902,1906,1921,1940,1964,1968,1981,1998,2018,2022,2035,2054],{"__ignoreMap":194},[198,1767,1768],{"class":77,"line":200},[198,1769,1770],{"class":207},"\"\"\"Compare a real pipeline in both libraries.\"\"\"\n",[198,1772,1773,1775],{"class":77,"line":218},[198,1774,244],{"class":243},[198,1776,1777],{"class":247}," time\n",[198,1779,1780],{"class":77,"line":263},[198,1781,260],{"emptyLinePlaceholder":259},[198,1783,1784,1786,1789,1791],{"class":77,"line":284},[198,1785,244],{"class":243},[198,1787,1788],{"class":247}," pandas ",[198,1790,251],{"class":243},[198,1792,1793],{"class":247}," pd\n",[198,1795,1796,1798,1800,1802],{"class":77,"line":294},[198,1797,244],{"class":243},[198,1799,248],{"class":247},[198,1801,251],{"class":243},[198,1803,254],{"class":247},[198,1805,1806],{"class":77,"line":540},[198,1807,260],{"emptyLinePlaceholder":259},[198,1809,1810,1813,1817],{"class":77,"line":560},[198,1811,1812],{"class":243},"def",[198,1814,1816],{"class":1815},"s_Opv"," timed",[198,1818,1819],{"class":247},"(label, fn):\n",[198,1821,1822,1825,1827],{"class":77,"line":566},[198,1823,1824],{"class":247},"    start ",[198,1826,269],{"class":243},[198,1828,1829],{"class":247}," time.perf_counter()\n",[198,1831,1832,1835,1837],{"class":77,"line":589},[198,1833,1834],{"class":247},"    result ",[198,1836,269],{"class":243},[198,1838,1839],{"class":247}," fn()\n",[198,1841,1842,1845,1847,1849,1851,1853,1856,1859,1861,1864,1867,1870,1873,1876,1878,1881,1883,1885,1888,1890,1892],{"class":77,"line":611},[198,1843,1844],{"class":287},"    print",[198,1846,399],{"class":247},[198,1848,1673],{"class":243},[198,1850,1676],{"class":207},[198,1852,440],{"class":1679},[198,1854,1855],{"class":247},"label",[198,1857,1858],{"class":243},":24s",[198,1860,1688],{"class":1679},[198,1862,1863],{"class":1679}," {",[198,1865,1866],{"class":247},"time.perf_counter() ",[198,1868,1869],{"class":243},"-",[198,1871,1872],{"class":247}," start",[198,1874,1875],{"class":243},":6.2f",[198,1877,1688],{"class":1679},[198,1879,1880],{"class":207},"s  rows=",[198,1882,440],{"class":1679},[198,1884,1682],{"class":287},[198,1886,1887],{"class":247},"(result)",[198,1889,1688],{"class":1679},[198,1891,1676],{"class":207},[198,1893,343],{"class":247},[198,1895,1896,1899],{"class":77,"line":617},[198,1897,1898],{"class":243},"    return",[198,1900,1901],{"class":247}," result\n",[198,1903,1904],{"class":77,"line":633},[198,1905,260],{"emptyLinePlaceholder":259},[198,1907,1908,1911,1914,1916,1918],{"class":77,"line":639},[198,1909,1910],{"class":247},"timed(",[198,1912,1913],{"class":207},"\"pandas openpyxl\"",[198,1915,331],{"class":247},[198,1917,1477],{"class":243},[198,1919,1920],{"class":247},": (\n",[198,1922,1923,1926,1928,1930,1933,1935,1938],{"class":77,"line":650},[198,1924,1925],{"class":247},"    pd.read_excel(",[198,1927,275],{"class":207},[198,1929,331],{"class":247},[198,1931,1932],{"class":334},"engine",[198,1934,269],{"class":243},[198,1936,1937],{"class":207},"\"openpyxl\"",[198,1939,343],{"class":247},[198,1941,1942,1945,1947,1949,1952,1954,1956,1959,1961],{"class":77,"line":669},[198,1943,1944],{"class":247},"      .groupby(",[198,1946,623],{"class":207},[198,1948,331],{"class":247},[198,1950,1951],{"class":334},"as_index",[198,1953,269],{"class":243},[198,1955,1445],{"class":287},[198,1957,1958],{"class":247},")[",[198,1960,546],{"class":207},[198,1962,1963],{"class":247},"].sum()\n",[198,1965,1966],{"class":77,"line":685},[198,1967,939],{"class":247},[198,1969,1970,1972,1975,1977,1979],{"class":77,"line":690},[198,1971,1910],{"class":247},[198,1973,1974],{"class":207},"\"pandas calamine\"",[198,1976,331],{"class":247},[198,1978,1477],{"class":243},[198,1980,1920],{"class":247},[198,1982,1983,1985,1987,1989,1991,1993,1996],{"class":77,"line":704},[198,1984,1925],{"class":247},[198,1986,275],{"class":207},[198,1988,331],{"class":247},[198,1990,1932],{"class":334},[198,1992,269],{"class":243},[198,1994,1995],{"class":207},"\"calamine\"",[198,1997,343],{"class":247},[198,1999,2000,2002,2004,2006,2008,2010,2012,2014,2016],{"class":77,"line":709},[198,2001,1944],{"class":247},[198,2003,623],{"class":207},[198,2005,331],{"class":247},[198,2007,1951],{"class":334},[198,2009,269],{"class":243},[198,2011,1445],{"class":287},[198,2013,1958],{"class":247},[198,2015,546],{"class":207},[198,2017,1963],{"class":247},[198,2019,2020],{"class":77,"line":1450},[198,2021,939],{"class":247},[198,2023,2024,2026,2029,2031,2033],{"class":77,"line":1460},[198,2025,1910],{"class":247},[198,2027,2028],{"class":207},"\"polars\"",[198,2030,331],{"class":247},[198,2032,1477],{"class":243},[198,2034,1920],{"class":247},[198,2036,2037,2040,2042,2045,2047,2050,2052],{"class":77,"line":1465},[198,2038,2039],{"class":247},"    pl.read_excel(",[198,2041,275],{"class":207},[198,2043,2044],{"class":247},").group_by(",[198,2046,623],{"class":207},[198,2048,2049],{"class":247},").agg(pl.col(",[198,2051,546],{"class":207},[198,2053,1009],{"class":247},[198,2055,2056],{"class":77,"line":1471},[198,2057,939],{"class":247},[25,2059,34,2064,34,2067,34,2070,34,2073,34,2076,34,2081,34,2086,34,2089,34,2093,34,2097,34,2100,34,2104,34,2107,34,2110,34,2114,34,2120,34,2122,34,2125,34,2128,34,2132,34,2135],{"viewBox":2060,"role":28,"ariaLabelledBy":2061,"xmlns":32,"style":33},"0 0 760 262",[2062,2063],"plx3-t","plx3-d",[36,2065,2066],{"id":2062},"Where the time goes in a read-and-aggregate job",[40,2068,2069],{"id":2063},"Parsing dominates a pure-Python read, so swapping to the calamine parser shortens the first bar; moving the aggregation to a multi-threaded engine shortens the second.",[44,2071],{"x":46,"y":46,"width":47,"height":2072,"fill":49},"262",[51,2074,2075],{"x":53,"y":54,"style":55},"Two separate costs, two separate fixes",[51,2077,2080],{"x":142,"y":2078,"style":2079},"62","font-size:12px;font-weight:700;fill:var(--text,#172033);text-anchor:start","pandas + openpyxl",[44,2082],{"x":142,"y":61,"width":2083,"height":783,"rx":2084,"fill":2085,"stroke":64},"440","6","#fee8f2",[44,2087],{"x":2088,"y":61,"width":60,"height":783,"rx":2084,"fill":93,"stroke":64},"500",[51,2090,2092],{"x":142,"y":2091,"style":2079},"128","pandas + calamine",[44,2094],{"x":142,"y":2095,"width":2096,"height":783,"rx":2084,"fill":112,"stroke":64},"136","170",[44,2098],{"x":2099,"y":2095,"width":60,"height":783,"rx":2084,"fill":93,"stroke":64},"230",[51,2101,2103],{"x":142,"y":2102,"style":2079},"194","polars",[44,2105],{"x":142,"y":2106,"width":2096,"height":783,"rx":2084,"fill":112,"stroke":64},"202",[44,2108],{"x":2099,"y":2106,"width":2109,"height":783,"rx":2084,"fill":144,"stroke":64},"54",[44,2111],{"x":142,"y":2112,"width":2113,"height":62,"rx":1319,"fill":2085,"stroke":64},"240","16",[51,2115,2119],{"x":2116,"y":2117,"style":2118},"84","251","font-size:11.5px;fill:var(--muted,#5b6780);text-anchor:start","pure-Python parse",[44,2121],{"x":91,"y":2112,"width":2113,"height":62,"rx":1319,"fill":112,"stroke":64},[51,2123,2124],{"x":831,"y":2117,"style":2118},"Rust parse",[44,2126],{"x":2127,"y":2112,"width":2113,"height":62,"rx":1319,"fill":93,"stroke":64},"316",[51,2129,2131],{"x":2130,"y":2117,"style":2118},"340","single-threaded aggregate",[44,2133],{"x":2134,"y":2112,"width":2113,"height":62,"rx":1319,"fill":144,"stroke":64},"524",[51,2136,2138],{"x":2137,"y":2117,"style":2118},"548","parallel aggregate",[10,2140,2141],{},"Run it three times and take the best, because the first read of a file is dominated by disk cache. If the parse dominates and the aggregation is trivial, switching engines inside pandas is the cheaper change; if the aggregation dominates, Polars is where the win is.",[181,2143,2145],{"id":2144},"read-a-folder-of-workbooks-into-one-frame","Read a folder of workbooks into one frame",[10,2147,2148],{},"Monthly exports arrive as one file per period, and the first job is nearly always to stack them. Polars concatenates with an explicit strategy for mismatched columns, which matters because a producer that adds a column in March should not silently drop the other months' data:",[189,2150,2152],{"className":234,"code":2151,"language":236,"meta":194,"style":194},"from pathlib import Path\n\nimport polars as pl\n\nframes = []\nfor path in sorted(Path(\"monthly\").glob(\"*.xlsx\")):\n    frame = pl.read_excel(path).with_columns(\n        source_file=pl.lit(path.name),\n        period=pl.lit(path.stem[-7:]),        # e.g. \"2026-03\"\n    )\n    frames.append(frame)\n\ncombined = pl.concat(frames, how=\"diagonal_relaxed\")\nprint(combined.shape, combined.columns)\n",[14,2153,2154,2164,2168,2178,2182,2192,2212,2222,2232,2253,2257,2262,2266,2285],{"__ignoreMap":194},[198,2155,2156,2158,2160,2162],{"class":77,"line":200},[198,2157,868],{"class":243},[198,2159,871],{"class":247},[198,2161,244],{"class":243},[198,2163,876],{"class":247},[198,2165,2166],{"class":77,"line":218},[198,2167,260],{"emptyLinePlaceholder":259},[198,2169,2170,2172,2174,2176],{"class":77,"line":263},[198,2171,244],{"class":243},[198,2173,248],{"class":247},[198,2175,251],{"class":243},[198,2177,254],{"class":247},[198,2179,2180],{"class":77,"line":284},[198,2181,260],{"emptyLinePlaceholder":259},[198,2183,2184,2187,2189],{"class":77,"line":294},[198,2185,2186],{"class":247},"frames ",[198,2188,269],{"class":243},[198,2190,2191],{"class":247}," []\n",[198,2193,2194,2196,2198,2200,2202,2204,2206,2208,2210],{"class":77,"line":540},[198,2195,904],{"class":243},[198,2197,907],{"class":247},[198,2199,910],{"class":243},[198,2201,913],{"class":287},[198,2203,916],{"class":247},[198,2205,919],{"class":207},[198,2207,922],{"class":247},[198,2209,925],{"class":207},[198,2211,928],{"class":247},[198,2213,2214,2217,2219],{"class":77,"line":560},[198,2215,2216],{"class":247},"    frame ",[198,2218,269],{"class":243},[198,2220,2221],{"class":247}," pl.read_excel(path).with_columns(\n",[198,2223,2224,2227,2229],{"class":77,"line":566},[198,2225,2226],{"class":334},"        source_file",[198,2228,269],{"class":243},[198,2230,2231],{"class":247},"pl.lit(path.name),\n",[198,2233,2234,2237,2239,2242,2244,2247,2250],{"class":77,"line":589},[198,2235,2236],{"class":334},"        period",[198,2238,269],{"class":243},[198,2240,2241],{"class":247},"pl.lit(path.stem[",[198,2243,1869],{"class":243},[198,2245,2246],{"class":287},"7",[198,2248,2249],{"class":247},":]),        ",[198,2251,2252],{"class":214},"# e.g. \"2026-03\"\n",[198,2254,2255],{"class":77,"line":611},[198,2256,1398],{"class":247},[198,2258,2259],{"class":77,"line":617},[198,2260,2261],{"class":247},"    frames.append(frame)\n",[198,2263,2264],{"class":77,"line":633},[198,2265,260],{"emptyLinePlaceholder":259},[198,2267,2268,2271,2273,2276,2278,2280,2283],{"class":77,"line":639},[198,2269,2270],{"class":247},"combined ",[198,2272,269],{"class":243},[198,2274,2275],{"class":247}," pl.concat(frames, ",[198,2277,1626],{"class":334},[198,2279,269],{"class":243},[198,2281,2282],{"class":207},"\"diagonal_relaxed\"",[198,2284,343],{"class":247},[198,2286,2287,2289],{"class":77,"line":650},[198,2288,288],{"class":287},[198,2290,2291],{"class":247},"(combined.shape, combined.columns)\n",[10,2293,2294,2297,2298,2301,2302,485],{},[14,2295,2296],{},"how=\"diagonal_relaxed\""," unions the columns, filling absent ones with nulls and widening types where they disagree — the behaviour you want for real exports. The stricter ",[14,2299,2300],{},"how=\"vertical\""," raises instead, which is the right choice when a schema change should stop the job. Tagging each row with its source file costs nothing and makes every later \"where did this number come from?\" question answerable; the pandas version of the same pattern is in ",[19,2303,2305],{"href":2304},"\u002Fgetting-started-with-python-excel-automation\u002Fworking-with-multiple-excel-sheets-in-python\u002Fcombine-multiple-excel-files-into-one-python\u002F","Combine multiple Excel files into one with Python",[181,2307,2309],{"id":2308},"fit-it-into-a-scheduled-report-job","Fit it into a scheduled report job",[10,2311,2312],{},"Nothing about Polars changes the shape of an unattended job — it slots into the same ingest, transform, generate, deliver pipeline. What it does change is where the time goes, which affects how you structure the schedule:",[189,2314,2316],{"className":234,"code":2315,"language":236,"meta":194,"style":194},"\"\"\"A nightly job: convert, aggregate, write, hand off for delivery.\"\"\"\nfrom pathlib import Path\n\nimport polars as pl\n\nRAW = Path(\"\u002Fsrv\u002Freports\u002Fincoming\")\nCACHE = Path(\"\u002Fsrv\u002Freports\u002Fcache\")\n\ndef refresh_cache() -> None:\n    \"\"\"Parse each new workbook once; every later read is columnar.\"\"\"\n    CACHE.mkdir(exist_ok=True)\n    for src in RAW.glob(\"*.xlsx\"):\n        dst = CACHE \u002F f\"{src.stem}.parquet\"\n        if not dst.exists() or dst.stat().st_mtime \u003C src.stat().st_mtime:\n            pl.read_excel(src).write_parquet(dst)\n\ndef build_summary() -> pl.DataFrame:\n    return (\n        pl.scan_parquet(CACHE \u002F \"*.parquet\")\n          .group_by(\"region\", \"product\")\n          .agg(revenue=pl.col(\"revenue\").sum())\n          .sort(\"revenue\", descending=True)\n          .collect()\n    )\n\nrefresh_cache()\nbuild_summary().write_excel(\"regional_summary.xlsx\", autofit=True)\n",[14,2317,2318,2323,2333,2337,2347,2351,2367,2381,2385,2400,2405,2422,2443,2471,2494,2499,2503,2513,2519,2533,2546,2561,2578,2583,2587,2591,2597],{"__ignoreMap":194},[198,2319,2320],{"class":77,"line":200},[198,2321,2322],{"class":207},"\"\"\"A nightly job: convert, aggregate, write, hand off for delivery.\"\"\"\n",[198,2324,2325,2327,2329,2331],{"class":77,"line":218},[198,2326,868],{"class":243},[198,2328,871],{"class":247},[198,2330,244],{"class":243},[198,2332,876],{"class":247},[198,2334,2335],{"class":77,"line":263},[198,2336,260],{"emptyLinePlaceholder":259},[198,2338,2339,2341,2343,2345],{"class":77,"line":284},[198,2340,244],{"class":243},[198,2342,248],{"class":247},[198,2344,251],{"class":243},[198,2346,254],{"class":247},[198,2348,2349],{"class":77,"line":294},[198,2350,260],{"emptyLinePlaceholder":259},[198,2352,2353,2356,2359,2362,2365],{"class":77,"line":540},[198,2354,2355],{"class":287},"RAW",[198,2357,2358],{"class":243}," =",[198,2360,2361],{"class":247}," Path(",[198,2363,2364],{"class":207},"\"\u002Fsrv\u002Freports\u002Fincoming\"",[198,2366,343],{"class":247},[198,2368,2369,2372,2374,2376,2379],{"class":77,"line":560},[198,2370,2371],{"class":287},"CACHE",[198,2373,2358],{"class":243},[198,2375,2361],{"class":247},[198,2377,2378],{"class":207},"\"\u002Fsrv\u002Freports\u002Fcache\"",[198,2380,343],{"class":247},[198,2382,2383],{"class":77,"line":566},[198,2384,260],{"emptyLinePlaceholder":259},[198,2386,2387,2389,2392,2395,2397],{"class":77,"line":589},[198,2388,1812],{"class":243},[198,2390,2391],{"class":1815}," refresh_cache",[198,2393,2394],{"class":247},"() -> ",[198,2396,386],{"class":287},[198,2398,2399],{"class":247},":\n",[198,2401,2402],{"class":77,"line":611},[198,2403,2404],{"class":207},"    \"\"\"Parse each new workbook once; every later read is columnar.\"\"\"\n",[198,2406,2407,2410,2413,2416,2418,2420],{"class":77,"line":617},[198,2408,2409],{"class":287},"    CACHE",[198,2411,2412],{"class":247},".mkdir(",[198,2414,2415],{"class":334},"exist_ok",[198,2417,269],{"class":243},[198,2419,1025],{"class":287},[198,2421,343],{"class":247},[198,2423,2424,2427,2430,2432,2435,2438,2440],{"class":77,"line":633},[198,2425,2426],{"class":243},"    for",[198,2428,2429],{"class":247}," src ",[198,2431,910],{"class":243},[198,2433,2434],{"class":287}," RAW",[198,2436,2437],{"class":247},".glob(",[198,2439,925],{"class":207},[198,2441,2442],{"class":247},"):\n",[198,2444,2445,2448,2450,2453,2456,2459,2461,2463,2466,2468],{"class":77,"line":639},[198,2446,2447],{"class":247},"        dst ",[198,2449,269],{"class":243},[198,2451,2452],{"class":287}," CACHE",[198,2454,2455],{"class":243}," \u002F",[198,2457,2458],{"class":243}," f",[198,2460,1676],{"class":207},[198,2462,440],{"class":1679},[198,2464,2465],{"class":247},"src.stem",[198,2467,1688],{"class":1679},[198,2469,2470],{"class":207},".parquet\"\n",[198,2472,2473,2476,2479,2482,2485,2488,2491],{"class":77,"line":650},[198,2474,2475],{"class":243},"        if",[198,2477,2478],{"class":243}," not",[198,2480,2481],{"class":247}," dst.exists() ",[198,2483,2484],{"class":243},"or",[198,2486,2487],{"class":247}," dst.stat().st_mtime ",[198,2489,2490],{"class":243},"\u003C",[198,2492,2493],{"class":247}," src.stat().st_mtime:\n",[198,2495,2496],{"class":77,"line":669},[198,2497,2498],{"class":247},"            pl.read_excel(src).write_parquet(dst)\n",[198,2500,2501],{"class":77,"line":685},[198,2502,260],{"emptyLinePlaceholder":259},[198,2504,2505,2507,2510],{"class":77,"line":690},[198,2506,1812],{"class":243},[198,2508,2509],{"class":1815}," build_summary",[198,2511,2512],{"class":247},"() -> pl.DataFrame:\n",[198,2514,2515,2517],{"class":77,"line":704},[198,2516,1898],{"class":243},[198,2518,537],{"class":247},[198,2520,2521,2524,2526,2528,2531],{"class":77,"line":709},[198,2522,2523],{"class":247},"        pl.scan_parquet(",[198,2525,2371],{"class":287},[198,2527,2455],{"class":243},[198,2529,2530],{"class":207}," \"*.parquet\"",[198,2532,343],{"class":247},[198,2534,2535,2538,2540,2542,2544],{"class":77,"line":1450},[198,2536,2537],{"class":247},"          .group_by(",[198,2539,623],{"class":207},[198,2541,331],{"class":247},[198,2543,679],{"class":207},[198,2545,343],{"class":247},[198,2547,2548,2551,2553,2555,2557,2559],{"class":77,"line":1460},[198,2549,2550],{"class":247},"          .agg(",[198,2552,1000],{"class":334},[198,2554,269],{"class":243},[198,2556,574],{"class":247},[198,2558,546],{"class":207},[198,2560,1009],{"class":247},[198,2562,2563,2566,2568,2570,2572,2574,2576],{"class":77,"line":1465},[198,2564,2565],{"class":247},"          .sort(",[198,2567,546],{"class":207},[198,2569,331],{"class":247},[198,2571,1020],{"class":334},[198,2573,269],{"class":243},[198,2575,1025],{"class":287},[198,2577,343],{"class":247},[198,2579,2580],{"class":77,"line":1471},[198,2581,2582],{"class":247},"          .collect()\n",[198,2584,2585],{"class":77,"line":1493},[198,2586,1398],{"class":247},[198,2588,2589],{"class":77,"line":1498},[198,2590,260],{"emptyLinePlaceholder":259},[198,2592,2594],{"class":77,"line":2593},26,[198,2595,2596],{"class":247},"refresh_cache()\n",[198,2598,2600,2603,2606,2608,2611,2613,2615],{"class":77,"line":2599},27,[198,2601,2602],{"class":247},"build_summary().write_excel(",[198,2604,2605],{"class":207},"\"regional_summary.xlsx\"",[198,2607,331],{"class":247},[198,2609,2610],{"class":334},"autofit",[198,2612,269],{"class":243},[198,2614,1025],{"class":287},[198,2616,343],{"class":247},[10,2618,1740,2619,2622,2623,1056,2627,485],{},[14,2620,2621],{},"mtime"," comparison means a rerun costs nothing for files that have not changed, so a job that is retried after a failure does not re-parse a whole archive. Delivery, logging and retries around this core are unchanged from any other report — see ",[19,2624,2626],{"href":2625},"\u002Fautomating-reporting-workflows\u002F","Automating Reporting Workflows",[19,2628,2630],{"href":2629},"\u002Fautomating-reporting-workflows\u002Ferror-handling-and-logging-in-excel-automation\u002Fretry-a-failed-excel-report-job-in-python\u002F","Retry a failed Excel report job in Python",[181,2632,2634],{"id":2633},"watch-the-differences-that-catch-pandas-users-out","Watch the differences that catch pandas users out",[10,2636,2637],{},"Most of the translation is mechanical, but four behaviours differ enough to cause a wrong number rather than an error:",[2639,2640,2641,2656,2681,2694],"ul",{},[2642,2643,2644,2648,2649,331,2652,2655],"li",{},[2645,2646,2647],"strong",{},"No index."," Polars has no row index at all. Anything you did with ",[14,2650,2651],{},"set_index",[14,2653,2654],{},"reset_index"," or index alignment becomes an explicit column and an explicit join.",[2642,2657,2658,2661,2662,2665,2666,2669,2670,2673,2674,2676,2677,2680],{},[2645,2659,2660],{},"Nulls are not NaN."," Polars distinguishes a missing value (",[14,2663,2664],{},"null",") from the float ",[14,2667,2668],{},"NaN",". A column of floats can contain both, and ",[14,2671,2672],{},"is_null()"," does not match ",[14,2675,2668],{}," — use ",[14,2678,2679],{},"is_nan()"," for that. This is the single most common source of surprise when porting a cleaning step.",[2642,2682,2683,2686,2687,2690,2691,2693],{},[2645,2684,2685],{},"Strict types."," A column is one type. Where pandas would happily hold a mixture in an ",[14,2688,2689],{},"object"," column, Polars forces a decision at read time — which is a feature, but it means ",[14,2692,475],{}," earns its place in the read call.",[2642,2695,2696,2699,2700,2703,2704,331,2707,331,2710,2713,2714,2717],{},[2645,2697,2698],{},"Expressions are lazy even in eager mode."," ",[14,2701,2702],{},"pl.col(\"revenue\") * 0.8"," is a description, not a value; it only computes inside ",[14,2705,2706],{},"select",[14,2708,2709],{},"with_columns",[14,2711,2712],{},"filter"," or ",[14,2715,2716],{},"agg",". Printing an expression shows the plan, not numbers.",[189,2719,2721],{"className":234,"code":2720,"language":236,"meta":194,"style":194},"import polars as pl\n\ndf = pl.DataFrame({\"value\": [1.0, None, float(\"nan\")]})\nprint(df.select(\n    nulls=pl.col(\"value\").is_null().sum(),\n    nans=pl.col(\"value\").is_nan().sum(),\n))\n",[14,2722,2723,2733,2737,2772,2779,2793,2807],{"__ignoreMap":194},[198,2724,2725,2727,2729,2731],{"class":77,"line":200},[198,2726,244],{"class":243},[198,2728,248],{"class":247},[198,2730,251],{"class":243},[198,2732,254],{"class":247},[198,2734,2735],{"class":77,"line":218},[198,2736,260],{"emptyLinePlaceholder":259},[198,2738,2739,2741,2743,2746,2749,2752,2755,2757,2759,2761,2764,2766,2769],{"class":77,"line":263},[198,2740,266],{"class":247},[198,2742,269],{"class":243},[198,2744,2745],{"class":247}," pl.DataFrame({",[198,2747,2748],{"class":207},"\"value\"",[198,2750,2751],{"class":247},": [",[198,2753,2754],{"class":287},"1.0",[198,2756,331],{"class":247},[198,2758,386],{"class":287},[198,2760,331],{"class":247},[198,2762,2763],{"class":287},"float",[198,2765,399],{"class":247},[198,2767,2768],{"class":207},"\"nan\"",[198,2770,2771],{"class":247},")]})\n",[198,2773,2774,2776],{"class":77,"line":284},[198,2775,288],{"class":287},[198,2777,2778],{"class":247},"(df.select(\n",[198,2780,2781,2784,2786,2788,2790],{"class":77,"line":294},[198,2782,2783],{"class":334},"    nulls",[198,2785,269],{"class":243},[198,2787,574],{"class":247},[198,2789,2748],{"class":207},[198,2791,2792],{"class":247},").is_null().sum(),\n",[198,2794,2795,2798,2800,2802,2804],{"class":77,"line":540},[198,2796,2797],{"class":334},"    nans",[198,2799,269],{"class":243},[198,2801,574],{"class":247},[198,2803,2748],{"class":207},[198,2805,2806],{"class":247},").is_nan().sum(),\n",[198,2808,2809],{"class":77,"line":560},[198,2810,939],{"class":247},[189,2812,2815],{"className":2813,"code":2814,"language":51,"meta":194},[303],"shape: (1, 2)\n┌───────┬──────┐\n│ nulls ┆ nans │\n│ 1     ┆ 1    │\n└───────┴──────┘\n",[14,2816,2814],{"__ignoreMap":194},[10,2818,2819,2820,2823,2824,2826,2827,485],{},"Knowing that distinction up front prevents the classic port bug: a ",[14,2821,2822],{},"fill_null(0)"," that leaves every ",[14,2825,2668],{}," untouched, and a total that quietly comes out wrong. The equivalent decisions on the pandas side are covered in ",[19,2828,2830],{"href":2829},"\u002Fadvanced-data-transformation-and-cleaning\u002Fhandling-missing-data-in-excel-reports\u002F","Handling Missing Data in Excel Reports",[181,2832,2834],{"id":2833},"what-to-install-and-what-each-package-is-for","What to install, and what each package is for",[10,2836,2837],{},"The dependency set is small, but each package has one job, and knowing which is which shortens the next debugging session:",[1181,2839,2840,2853],{},[1184,2841,2842],{},[1187,2843,2844,2847,2850],{},[1190,2845,2846],{},"Package",[1190,2848,2849],{},"Provides",[1190,2851,2852],{},"Needed when",[1199,2854,2855,2867,2883,2898,2911],{},[1187,2856,2857,2861,2864],{},[1204,2858,2859],{},[14,2860,2103],{},[1204,2862,2863],{},"The DataFrame library and query engine",[1204,2865,2866],{},"Always",[1187,2868,2869,2874,2877],{},[1204,2870,2871],{},[14,2872,2873],{},"fastexcel",[1204,2875,2876],{},"Python bindings over the calamine reader",[1204,2878,2879,2882],{},[14,2880,2881],{},"pl.read_excel()"," on any format",[1187,2884,2885,2890,2895],{},[1204,2886,2887],{},[14,2888,2889],{},"xlsxwriter",[1204,2891,2892,2893],{},"The writer beneath ",[14,2894,150],{},[1204,2896,2897],{},"Producing a formatted workbook",[1187,2899,2900,2905,2908],{},[1204,2901,2902],{},[14,2903,2904],{},"pyarrow",[1204,2906,2907],{},"Arrow interchange and Parquet support",[1204,2909,2910],{},"Converting to pandas, reading or writing Parquet",[1187,2912,2913,2918,2921],{},[1204,2914,2915],{},[14,2916,2917],{},"connectorx",[1204,2919,2920],{},"Fast database reads",[1204,2922,2923,2926],{},[14,2924,2925],{},"pl.read_database_uri()"," against SQL",[10,2928,2929,2930,2933,2934,2937],{},"Installing ",[14,2931,2932],{},"\"polars[excel]\""," pulls the reader; ",[14,2935,2936],{},"\"polars[all]\""," pulls everything above and is convenient in a container image where a missing extra means a failed nightly run rather than a quick local fix.",[181,2939,2941],{"id":2940},"key-takeaways","Key takeaways",[2639,2943,2944,2949,2955,2958,2961,2964],{},[2642,2945,2946,2948],{},[14,2947,2881],{}," parses through the Rust calamine reader, which is consistently faster and leaner than pure-Python engines.",[2642,2950,2951,2952,2954],{},"Set ",[14,2953,475],{}," at read time to control types instead of repairing them afterwards.",[2642,2956,2957],{},"Expressions are planned and executed across cores; the win grows with the size of the transformation, not just the size of the file.",[2642,2959,2960],{},"Polars and pandas convert through Arrow at negligible cost, so adoption can be incremental and reversible.",[2642,2962,2963],{},"Convert repeatedly-read workbooks to Parquet and scan them lazily — the largest speed-up available in most reporting jobs.",[2642,2965,2966,2969],{},[14,2967,2968],{},"write_excel()"," covers formatted output; templates, in-place edits and cell-level formatting still belong to openpyxl.",[181,2971,2973],{"id":2972},"frequently-asked-questions","Frequently asked questions",[10,2975,2976,2979],{},[2645,2977,2978],{},"Is Polars faster than pandas for Excel files?","\nFor the parsing itself, usually yes, because Polars reads through calamine — a Rust reader — rather than a pure-Python parser. The larger win is in what follows: group-bys, joins and filters over Arrow-backed columns run multi-threaded, so a heavy transformation after the read is where the difference shows.",[10,2981,2982,2985],{},[2645,2983,2984],{},"Do I have to rewrite everything to adopt Polars?","\nNo. Polars and pandas convert to each other cheaply through Arrow, so a common pattern is to read and transform in Polars, then hand a pandas DataFrame to whatever formatting code you already have.",[10,2987,2988,2991,2993],{},[2645,2989,2990],{},"Can Polars write a styled Excel report?",[14,2992,150],{}," drives xlsxwriter under the hood, so it can apply number formats, autofit columns, add an Excel table and even a conditional format. For fine-grained control over an existing workbook you still want openpyxl.",[10,2995,2996,2999,3000,2713,3002,3004,3005,3007],{},[2645,2997,2998],{},"Does Polars handle multiple sheets?","\nYes. Pass ",[14,3001,335],{},[14,3003,359],{},", or ",[14,3006,386],{}," to get a dictionary of DataFrames keyed by sheet name, in the same shape pandas returns.",[10,3009,3010,3013],{},[2645,3011,3012],{},"When should I convert Excel data to Parquet?","\nAs soon as the same workbook is read more than once. Parquet keeps types, compresses well, and loads an order of magnitude faster, so a nightly conversion turns a slow Excel read into a fast columnar read for every downstream job.",[181,3015,3017],{"id":3016},"conclusion","Conclusion",[10,3019,3020],{},"Polars gives Excel automation a faster front end without asking you to abandon the tools that make workbooks presentable. Read through calamine, control types at the boundary, express transformations as column operations, and convert anything you will read twice to Parquet. Then hand the result to xlsxwriter or openpyxl for the formatting, charts and templates that only a spreadsheet library can produce.",[181,3022,3024],{"id":3023},"related","Related",[2639,3026,3027,3035,3042,3049,3059,3064,3078],{},[2642,3028,3029,2699,3032,3034],{},[2645,3030,3031],{},"Up:",[19,3033,22],{"href":21}," — the section this topic extends, and where the pandas equivalents live.",[2642,3036,3037,3041],{},[19,3038,3040],{"href":3039},"\u002Fadvanced-data-transformation-and-cleaning\u002Freading-excel-with-polars-and-arrow\u002Fread-an-excel-file-with-polars-read-excel\u002F","Read an Excel file with polars.read_excel"," — the reader in full, including sheets, headers and schema control.",[2642,3043,3044,3048],{},[19,3045,3047],{"href":3046},"\u002Fadvanced-data-transformation-and-cleaning\u002Freading-excel-with-polars-and-arrow\u002Fspeed-up-pandas-excel-reads-with-the-calamine-engine\u002F","Speed up pandas Excel reads with the calamine engine"," — the same fast parser without leaving pandas.",[2642,3050,3051,3055,3056,3058],{},[19,3052,3054],{"href":3053},"\u002Fadvanced-data-transformation-and-cleaning\u002Freading-excel-with-polars-and-arrow\u002Fwrite-a-polars-dataframe-to-excel-with-formatting\u002F","Write a Polars DataFrame to Excel with formatting"," — ",[14,3057,150],{}," options for a delivered report.",[2642,3060,3061,3063],{},[19,3062,1063],{"href":1062}," — the conversion that makes every later read cheap.",[2642,3065,3066,2699,3069,1056,3073,3077],{},[2645,3067,3068],{},"Sibling topics:",[19,3070,3072],{"href":3071},"\u002Fadvanced-data-transformation-and-cleaning\u002Fworking-with-large-excel-files-in-python\u002F","Working with Large Excel Files in Python",[19,3074,3076],{"href":3075},"\u002Fadvanced-data-transformation-and-cleaning\u002Fcleaning-excel-data-with-pandas\u002F","Cleaning Excel Data with pandas"," — the pandas-side techniques these tools complement.",[2642,3079,3080,3084],{},[19,3081,3083],{"href":3082},"\u002Fgetting-started-with-python-excel-automation\u002Fchoosing-a-python-excel-library\u002Fpandas-vs-polars-for-excel-workflows\u002F","pandas vs Polars for Excel Workflows"," — where the two genuinely differ, and how to move between them.",[3086,3087,3088],"style",{},"html pre.shiki code .sMTad, html code.shiki .sMTad{--shiki-default:#6F42C1;--shiki-dark:#FFB757}html pre.shiki code .srMev, html code.shiki .srMev{--shiki-default:#032F62;--shiki-dark:#ADDCFF}html pre.shiki code .s-wDw, html code.shiki .s-wDw{--shiki-default:#6A737D;--shiki-dark:#BDC4CC}html .default .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .dark .shiki span {color: var(--shiki-dark);background: var(--shiki-dark-bg);font-style: var(--shiki-dark-font-style);font-weight: var(--shiki-dark-font-weight);text-decoration: var(--shiki-dark-text-decoration);}html.dark .shiki span {color: var(--shiki-dark);background: var(--shiki-dark-bg);font-style: var(--shiki-dark-font-style);font-weight: var(--shiki-dark-font-weight);text-decoration: var(--shiki-dark-text-decoration);}html pre.shiki code .s-kum, html code.shiki .s-kum{--shiki-default:#D73A49;--shiki-dark:#FF9492}html pre.shiki code .skGVy, html code.shiki .skGVy{--shiki-default:#24292E;--shiki-dark:#F0F3F6}html pre.shiki code .sP0c6, html code.shiki .sP0c6{--shiki-default:#005CC5;--shiki-dark:#91CBFF}html pre.shiki code .sa561, html code.shiki .sa561{--shiki-default:#E36209;--shiki-dark:#FFB757}html pre.shiki code .sSjpA, html code.shiki .sSjpA{--shiki-default:#005CC5;--shiki-dark:#FF9492}html pre.shiki code .s_Opv, html code.shiki .s_Opv{--shiki-default:#6F42C1;--shiki-dark:#DBB7FF}",{"title":194,"searchDepth":218,"depth":218,"links":3090},[3091,3092,3093,3094,3095,3096,3097,3098,3099,3100,3101,3102,3103,3104,3105,3106,3107],{"id":183,"depth":218,"text":184},{"id":488,"depth":218,"text":489},{"id":720,"depth":218,"text":721},{"id":854,"depth":218,"text":855},{"id":1066,"depth":218,"text":1067},{"id":1175,"depth":218,"text":1176},{"id":1268,"depth":218,"text":1269},{"id":1525,"depth":218,"text":1526},{"id":1756,"depth":218,"text":1757},{"id":2144,"depth":218,"text":2145},{"id":2308,"depth":218,"text":2309},{"id":2633,"depth":218,"text":2634},{"id":2833,"depth":218,"text":2834},{"id":2940,"depth":218,"text":2941},{"id":2972,"depth":218,"text":2973},{"id":3016,"depth":218,"text":3017},{"id":3023,"depth":218,"text":3024},"2026-09-05","2026-08-27","Read and write Excel with Polars and Arrow-backed engines: polars.read_excel, the calamine reader, lazy pipelines, Parquet conversion, and when pandas is still the better tool.","md",[3113,3117,3119,3121,3123],{"q":2978,"a":3114},{"For the parsing itself, usually yes, because Polars reads through calamine — a Rust reader — rather than a pure-Python parser":3115},{" The larger win is in what follows":3116},"group-bys, joins and filters over Arrow-backed columns run multi-threaded, so a heavy transformation after the read is where the difference shows.",{"q":2984,"a":3118},"No. Polars and pandas convert to each other cheaply through Arrow, so a common pattern is to read and transform in Polars, then hand a pandas DataFrame to whatever formatting code you already have.",{"q":2990,"a":3120},"write_excel drives xlsxwriter under the hood, so it can apply number formats, autofit columns, add an Excel table and even a conditional format. For fine-grained control over an existing workbook you still want openpyxl.",{"q":2998,"a":3122},"Yes. Pass sheet_name or sheet_id, or None to get a dictionary of DataFrames keyed by sheet name, in the same shape pandas returns.",{"q":3012,"a":3124},"As soon as the same workbook is read more than once. Parquet keeps types, compresses well, and loads an order of magnitude faster, so a nightly conversion turns a slow Excel read into a fast columnar read for every downstream job.",{},"\u002Fadvanced-data-transformation-and-cleaning\u002Freading-excel-with-polars-and-arrow",{"title":3128,"description":3129},"Read Excel with Polars and Arrow","A practical guide to Polars for Excel work: read_excel with calamine, schema control, expression pipelines, writing formatted xlsx, and converting workbooks to Parquet.","reading-excel-with-polars-and-arrow","advanced-data-transformation-and-cleaning\u002Freading-excel-with-polars-and-arrow\u002Findex","guide","RwNR-42hhAROFDzGsxjtYbXGSg0jF9JKsY5WDm0XDFs",[3135,3139],{"title":3136,"path":3137,"stem":3138,"children":-1},"Sync Google Sheets with Excel Using Python","\u002Fadvanced-data-transformation-and-cleaning\u002Fmoving-data-between-excel-and-databases\u002Fsync-google-sheets-with-excel-using-python","advanced-data-transformation-and-cleaning\u002Fmoving-data-between-excel-and-databases\u002Fsync-google-sheets-with-excel-using-python\u002Findex",{"title":3140,"path":3141,"stem":3142,"children":-1},"Convert Excel Files to Parquet with Python","\u002Fadvanced-data-transformation-and-cleaning\u002Freading-excel-with-polars-and-arrow\u002Fconvert-excel-files-to-parquet-with-python","advanced-data-transformation-and-cleaning\u002Freading-excel-with-polars-and-arrow\u002Fconvert-excel-files-to-parquet-with-python\u002Findex",1788710151645]