[{"data":1,"prerenderedAt":2331},["ShallowReactive",2],{"doc:\u002Fadvanced-data-transformation-and-cleaning\u002Fworking-with-large-excel-files-in-python\u002Fprocess-multiple-excel-files-in-parallel-with-python":3,"surround:\u002Fadvanced-data-transformation-and-cleaning\u002Fworking-with-large-excel-files-in-python\u002Fprocess-multiple-excel-files-in-parallel-with-python":2323},{"id":4,"title":5,"body":6,"dateModified":2294,"datePublished":2294,"description":2295,"extension":2296,"faq":2297,"meta":2306,"navigation":253,"path":2316,"seo":2317,"slug":2319,"stem":2320,"type":2321,"__hash__":2322},"docs\u002Fadvanced-data-transformation-and-cleaning\u002Fworking-with-large-excel-files-in-python\u002Fprocess-multiple-excel-files-in-parallel-with-python\u002Findex.md","Process Multiple Excel Files in Parallel with Python",{"type":7,"value":8,"toc":2278},"minimark",[9,19,160,165,193,197,204,520,523,527,530,871,874,877,881,1331,1342,1346,1461,1464,1468,1536,1543,1547,1550,1723,1726,1730,1733,1804,1807,1811,1926,1930,1936,2169,2180,2184,2187,2191,2194,2198,2208,2214,2220,2226,2230,2233,2242,2245,2274],[10,11,12,13,18],"p",{},"A folder of thirty monthly workbooks is the ideal shape for parallelism: the files are independent, each one is CPU-bound XML parsing, and the results are small. Done right this is close to a linear speed-up; done wrong it exhausts memory or hides a corrupt file. This guide, part of ",[14,15,17],"a",{"href":16},"\u002Fadvanced-data-transformation-and-cleaning\u002Fworking-with-large-excel-files-in-python\u002F","Working with Large Excel Files in Python",", covers the pattern and its limits.",[20,21,29,30,29,34,29,38,29,45,29,53,29,60,29,69,29,73,29,77,29,81,29,89,29,95,29,100,29,103,29,107,29,110,29,113,29,117,29,120,29,125,29,130,29,133,29,138,29,141,29,144,29,147,29,156],"svg",{"viewBox":22,"role":23,"ariaLabelledBy":24,"xmlns":27,"style":28},"0 0 740 248","img",[25,26],"par-shape-t","par-shape-d","http:\u002F\u002Fwww.w3.org\u002F2000\u002Fsvg","width:100%;max-width:740px;height:auto;display:block;margin:1.5rem auto;font-family:Inter,ui-sans-serif,system-ui,sans-serif","\n  ",[31,32,33],"title",{"id":25},"One worker process per file, small results back",[35,36,37],"desc",{"id":26},"The parent process lists the workbooks and submits one job per file. Four worker processes each parse their own workbook independently and return a small aggregate. The parent combines the aggregates into a single summary.",[39,40],"rect",{"x":41,"y":41,"width":42,"height":43,"fill":44},"0","740","248","#ffffff",[39,46],{"x":47,"y":48,"width":49,"height":50,"rx":51,"fill":52},"270","18","200","46","12","#5b5cf0",[54,55,59],"text",{"x":56,"y":57,"style":58},"370","47","font-size:12.5px;font-weight:700;fill:#ffffff;text-anchor:middle","parent: 30 files listed",[61,62],"line",{"x1":63,"y1":64,"x2":65,"y2":66,"stroke":67,"style":68},"310","66","110","98","var(--muted,#5b6780)","stroke-width:1.5px",[61,70],{"x1":71,"y1":64,"x2":72,"y2":66,"stroke":67,"style":68},"345","290",[61,74],{"x1":75,"y1":64,"x2":76,"y2":66,"stroke":67,"style":68},"395","450",[61,78],{"x1":79,"y1":64,"x2":80,"y2":66,"stroke":67,"style":68},"430","630",[39,82],{"x":83,"y":84,"width":85,"height":86,"rx":51,"fill":87,"stroke":88},"24","100","170","72","#f0f4ff","var(--brand,#5b5cf0)",[54,90,94],{"x":91,"y":92,"style":93},"109","128","font-size:12px;font-weight:700;fill:var(--brand-strong,#4338ca);text-anchor:middle","worker 1",[54,96,99],{"x":91,"y":97,"style":98},"152","font-size:11px;fill:var(--muted,#5b6780);text-anchor:middle","jan.xlsx → 3 numbers",[39,101],{"x":102,"y":84,"width":85,"height":86,"rx":51,"fill":87,"stroke":88},"204",[54,104,106],{"x":105,"y":92,"style":93},"289","worker 2",[54,108,109],{"x":105,"y":97,"style":98},"feb.xlsx → 3 numbers",[39,111],{"x":112,"y":84,"width":85,"height":86,"rx":51,"fill":87,"stroke":88},"384",[54,114,116],{"x":115,"y":92,"style":93},"469","worker 3",[54,118,119],{"x":115,"y":97,"style":98},"mar.xlsx → 3 numbers",[39,121],{"x":122,"y":84,"width":97,"height":86,"rx":51,"fill":123,"stroke":124},"564","#fce9e9","var(--danger,#dc2626)",[54,126,129],{"x":127,"y":92,"style":128},"640","font-size:12px;font-weight:700;fill:var(--danger,#dc2626);text-anchor:middle","worker 4",[54,131,132],{"x":127,"y":97,"style":98},"apr.xlsx → error",[61,134],{"x1":91,"y1":135,"x2":136,"y2":102,"stroke":137,"style":68},"174","330","var(--teal,#0f9488)",[61,139],{"x1":105,"y1":135,"x2":140,"y2":102,"stroke":137,"style":68},"350",[61,142],{"x1":115,"y1":135,"x2":143,"y2":102,"stroke":137,"style":68},"390",[61,145],{"x1":127,"y1":135,"x2":146,"y2":102,"stroke":124,"style":68},"410",[39,148],{"x":149,"y":150,"width":151,"height":152,"rx":153,"fill":154,"stroke":137,"style":155},"230","206","280","38","10","#d9f4f1","stroke-width:2px",[54,157,159],{"x":56,"y":149,"style":158},"font-size:12px;font-weight:700;fill:var(--teal-ink,#0b6157);text-anchor:middle","combined summary + one failure logged",[161,162,164],"h2",{"id":163},"prerequisites","Prerequisites",[166,167,172],"pre",{"className":168,"code":169,"language":170,"meta":171,"style":171},"language-bash shiki shiki-themes github-light github-dark-high-contrast","pip install pandas openpyxl\n","bash","",[173,174,175],"code",{"__ignoreMap":171},[176,177,179,183,187,190],"span",{"class":61,"line":178},1,[176,180,182],{"class":181},"sMTad","pip",[176,184,186],{"class":185},"srMev"," install",[176,188,189],{"class":185}," pandas",[176,191,192],{"class":185}," openpyxl\n",[161,194,196],{"id":195},"why-processes-not-threads","Why processes, not threads",[10,198,199,200,203],{},"Parsing an ",[173,201,202],{},".xlsx"," is Python bytecode doing XML work, so it holds the global interpreter lock almost continuously. Threads therefore interleave rather than overlap:",[166,205,209],{"className":206,"code":207,"language":208,"meta":171,"style":171},"language-python shiki shiki-themes github-light github-dark-high-contrast","import time\nfrom concurrent.futures import ProcessPoolExecutor, ThreadPoolExecutor\nfrom pathlib import Path\n\nimport pandas as pd\n\ndef row_count(path):\n    return len(pd.read_excel(path, sheet_name=0, usecols=[0]))\n\nfiles = sorted(str(p) for p in Path(\"monthly\").glob(\"*.xlsx\"))\n\nfor Pool, label in ((ThreadPoolExecutor, \"threads\"), (ProcessPoolExecutor, \"processes\")):\n    start = time.perf_counter()\n    with Pool(max_workers=4) as pool:\n        totals = list(pool.map(row_count, files))\n    print(f\"{label:10} {time.perf_counter() - start:5.1f}s  {sum(totals):,} rows\")\n","python",[173,210,211,221,235,248,255,269,274,287,325,330,374,379,404,415,440,454],{"__ignoreMap":171},[176,212,213,217],{"class":61,"line":178},[176,214,216],{"class":215},"s-kum","import",[176,218,220],{"class":219},"skGVy"," time\n",[176,222,224,227,230,232],{"class":61,"line":223},2,[176,225,226],{"class":215},"from",[176,228,229],{"class":219}," concurrent.futures ",[176,231,216],{"class":215},[176,233,234],{"class":219}," ProcessPoolExecutor, ThreadPoolExecutor\n",[176,236,238,240,243,245],{"class":61,"line":237},3,[176,239,226],{"class":215},[176,241,242],{"class":219}," pathlib ",[176,244,216],{"class":215},[176,246,247],{"class":219}," Path\n",[176,249,251],{"class":61,"line":250},4,[176,252,254],{"emptyLinePlaceholder":253},true,"\n",[176,256,258,260,263,266],{"class":61,"line":257},5,[176,259,216],{"class":215},[176,261,262],{"class":219}," pandas ",[176,264,265],{"class":215},"as",[176,267,268],{"class":219}," pd\n",[176,270,272],{"class":61,"line":271},6,[176,273,254],{"emptyLinePlaceholder":253},[176,275,277,280,284],{"class":61,"line":276},7,[176,278,279],{"class":215},"def",[176,281,283],{"class":282},"s_Opv"," row_count",[176,285,286],{"class":219},"(path):\n",[176,288,290,293,297,300,304,307,309,312,315,317,320,322],{"class":61,"line":289},8,[176,291,292],{"class":215},"    return",[176,294,296],{"class":295},"sP0c6"," len",[176,298,299],{"class":219},"(pd.read_excel(path, ",[176,301,303],{"class":302},"sa561","sheet_name",[176,305,306],{"class":215},"=",[176,308,41],{"class":295},[176,310,311],{"class":219},", ",[176,313,314],{"class":302},"usecols",[176,316,306],{"class":215},[176,318,319],{"class":219},"[",[176,321,41],{"class":295},[176,323,324],{"class":219},"]))\n",[176,326,328],{"class":61,"line":327},9,[176,329,254],{"emptyLinePlaceholder":253},[176,331,333,336,338,341,344,347,350,353,356,359,362,365,368,371],{"class":61,"line":332},10,[176,334,335],{"class":219},"files ",[176,337,306],{"class":215},[176,339,340],{"class":295}," sorted",[176,342,343],{"class":219},"(",[176,345,346],{"class":295},"str",[176,348,349],{"class":219},"(p) ",[176,351,352],{"class":215},"for",[176,354,355],{"class":219}," p ",[176,357,358],{"class":215},"in",[176,360,361],{"class":219}," Path(",[176,363,364],{"class":185},"\"monthly\"",[176,366,367],{"class":219},").glob(",[176,369,370],{"class":185},"\"*.xlsx\"",[176,372,373],{"class":219},"))\n",[176,375,377],{"class":61,"line":376},11,[176,378,254],{"emptyLinePlaceholder":253},[176,380,382,384,387,389,392,395,398,401],{"class":61,"line":381},12,[176,383,352],{"class":215},[176,385,386],{"class":219}," Pool, label ",[176,388,358],{"class":215},[176,390,391],{"class":219}," ((ThreadPoolExecutor, ",[176,393,394],{"class":185},"\"threads\"",[176,396,397],{"class":219},"), (ProcessPoolExecutor, ",[176,399,400],{"class":185},"\"processes\"",[176,402,403],{"class":219},")):\n",[176,405,407,410,412],{"class":61,"line":406},13,[176,408,409],{"class":219},"    start ",[176,411,306],{"class":215},[176,413,414],{"class":219}," time.perf_counter()\n",[176,416,418,421,424,427,429,432,435,437],{"class":61,"line":417},14,[176,419,420],{"class":215},"    with",[176,422,423],{"class":219}," Pool(",[176,425,426],{"class":302},"max_workers",[176,428,306],{"class":215},[176,430,431],{"class":295},"4",[176,433,434],{"class":219},") ",[176,436,265],{"class":215},[176,438,439],{"class":219}," pool:\n",[176,441,443,446,448,451],{"class":61,"line":442},15,[176,444,445],{"class":219},"        totals ",[176,447,306],{"class":215},[176,449,450],{"class":295}," list",[176,452,453],{"class":219},"(pool.map(row_count, files))\n",[176,455,457,460,462,465,468,472,475,478,481,484,487,490,493,496,498,501,503,506,509,512,514,517],{"class":61,"line":456},16,[176,458,459],{"class":295},"    print",[176,461,343],{"class":219},[176,463,464],{"class":215},"f",[176,466,467],{"class":185},"\"",[176,469,471],{"class":470},"sSjpA","{",[176,473,474],{"class":219},"label",[176,476,477],{"class":215},":10",[176,479,480],{"class":470},"}",[176,482,483],{"class":470}," {",[176,485,486],{"class":219},"time.perf_counter() ",[176,488,489],{"class":215},"-",[176,491,492],{"class":219}," start",[176,494,495],{"class":215},":5.1f",[176,497,480],{"class":470},[176,499,500],{"class":185},"s  ",[176,502,471],{"class":470},[176,504,505],{"class":295},"sum",[176,507,508],{"class":219},"(totals)",[176,510,511],{"class":215},":,",[176,513,480],{"class":470},[176,515,516],{"class":185}," rows\"",[176,518,519],{"class":219},")\n",[10,521,522],{},"On a machine with four or more cores the process version is typically two to three times faster; the thread version is often no faster than a plain loop. Threads still earn their place when the bottleneck is the network — downloading thirty workbooks from object storage is I\u002FO-bound and threads handle it well — so the rule is threads for fetching, processes for parsing.",[161,524,526],{"id":525},"a-worker-that-returns-something-small","A worker that returns something small",[10,528,529],{},"The unit of work is one file, and what it returns crosses a process boundary — so return an aggregate, not the data:",[166,531,533],{"className":206,"code":532,"language":208,"meta":171,"style":171},"from pathlib import Path\n\nimport pandas as pd\n\ndef summarise(path):\n    \"\"\"Runs in a worker process. Returns a small, picklable dict.\"\"\"\n    try:\n        df = pd.read_excel(path, sheet_name=\"Orders\",\n                           usecols=[\"Region\", \"Quantity\", \"Unit_Price\"])\n        df[\"Revenue\"] = df[\"Quantity\"] * df[\"Unit_Price\"]\n        by_region = df.groupby(\"Region\", observed=True)[\"Revenue\"].sum().round(2)\n        return {\n            \"file\": Path(path).name,\n            \"rows\": int(len(df)),\n            \"revenue\": float(df[\"Revenue\"].sum()),\n            \"by_region\": by_region.to_dict(),\n            \"error\": None,\n        }\n    except Exception as exc:                      # noqa: BLE001 - report, do not crash the batch\n        return {\"file\": Path(path).name, \"rows\": 0, \"revenue\": 0.0,\n                \"by_region\": {}, \"error\": f\"{type(exc).__name__}: {exc}\"}\n",[173,534,535,545,549,559,563,572,577,585,605,630,660,695,703,711,730,748,756,769,775,794,826],{"__ignoreMap":171},[176,536,537,539,541,543],{"class":61,"line":178},[176,538,226],{"class":215},[176,540,242],{"class":219},[176,542,216],{"class":215},[176,544,247],{"class":219},[176,546,547],{"class":61,"line":223},[176,548,254],{"emptyLinePlaceholder":253},[176,550,551,553,555,557],{"class":61,"line":237},[176,552,216],{"class":215},[176,554,262],{"class":219},[176,556,265],{"class":215},[176,558,268],{"class":219},[176,560,561],{"class":61,"line":250},[176,562,254],{"emptyLinePlaceholder":253},[176,564,565,567,570],{"class":61,"line":257},[176,566,279],{"class":215},[176,568,569],{"class":282}," summarise",[176,571,286],{"class":219},[176,573,574],{"class":61,"line":271},[176,575,576],{"class":185},"    \"\"\"Runs in a worker process. Returns a small, picklable dict.\"\"\"\n",[176,578,579,582],{"class":61,"line":276},[176,580,581],{"class":215},"    try",[176,583,584],{"class":219},":\n",[176,586,587,590,592,595,597,599,602],{"class":61,"line":289},[176,588,589],{"class":219},"        df ",[176,591,306],{"class":215},[176,593,594],{"class":219}," pd.read_excel(path, ",[176,596,303],{"class":302},[176,598,306],{"class":215},[176,600,601],{"class":185},"\"Orders\"",[176,603,604],{"class":219},",\n",[176,606,607,610,612,614,617,619,622,624,627],{"class":61,"line":327},[176,608,609],{"class":302},"                           usecols",[176,611,306],{"class":215},[176,613,319],{"class":219},[176,615,616],{"class":185},"\"Region\"",[176,618,311],{"class":219},[176,620,621],{"class":185},"\"Quantity\"",[176,623,311],{"class":219},[176,625,626],{"class":185},"\"Unit_Price\"",[176,628,629],{"class":219},"])\n",[176,631,632,635,638,641,643,646,648,650,653,655,657],{"class":61,"line":332},[176,633,634],{"class":219},"        df[",[176,636,637],{"class":185},"\"Revenue\"",[176,639,640],{"class":219},"] ",[176,642,306],{"class":215},[176,644,645],{"class":219}," df[",[176,647,621],{"class":185},[176,649,640],{"class":219},[176,651,652],{"class":215},"*",[176,654,645],{"class":219},[176,656,626],{"class":185},[176,658,659],{"class":219},"]\n",[176,661,662,665,667,670,672,674,677,679,682,685,687,690,693],{"class":61,"line":376},[176,663,664],{"class":219},"        by_region ",[176,666,306],{"class":215},[176,668,669],{"class":219}," df.groupby(",[176,671,616],{"class":185},[176,673,311],{"class":219},[176,675,676],{"class":302},"observed",[176,678,306],{"class":215},[176,680,681],{"class":295},"True",[176,683,684],{"class":219},")[",[176,686,637],{"class":185},[176,688,689],{"class":219},"].sum().round(",[176,691,692],{"class":295},"2",[176,694,519],{"class":219},[176,696,697,700],{"class":61,"line":381},[176,698,699],{"class":215},"        return",[176,701,702],{"class":219}," {\n",[176,704,705,708],{"class":61,"line":406},[176,706,707],{"class":185},"            \"file\"",[176,709,710],{"class":219},": Path(path).name,\n",[176,712,713,716,719,722,724,727],{"class":61,"line":417},[176,714,715],{"class":185},"            \"rows\"",[176,717,718],{"class":219},": ",[176,720,721],{"class":295},"int",[176,723,343],{"class":219},[176,725,726],{"class":295},"len",[176,728,729],{"class":219},"(df)),\n",[176,731,732,735,737,740,743,745],{"class":61,"line":442},[176,733,734],{"class":185},"            \"revenue\"",[176,736,718],{"class":219},[176,738,739],{"class":295},"float",[176,741,742],{"class":219},"(df[",[176,744,637],{"class":185},[176,746,747],{"class":219},"].sum()),\n",[176,749,750,753],{"class":61,"line":456},[176,751,752],{"class":185},"            \"by_region\"",[176,754,755],{"class":219},": by_region.to_dict(),\n",[176,757,759,762,764,767],{"class":61,"line":758},17,[176,760,761],{"class":185},"            \"error\"",[176,763,718],{"class":219},[176,765,766],{"class":295},"None",[176,768,604],{"class":219},[176,770,772],{"class":61,"line":771},18,[176,773,774],{"class":219},"        }\n",[176,776,778,781,784,787,790],{"class":61,"line":777},19,[176,779,780],{"class":215},"    except",[176,782,783],{"class":295}," Exception",[176,785,786],{"class":215}," as",[176,788,789],{"class":219}," exc:                      ",[176,791,793],{"class":792},"s-wDw","# noqa: BLE001 - report, do not crash the batch\n",[176,795,797,799,801,804,807,810,812,814,816,819,821,824],{"class":61,"line":796},20,[176,798,699],{"class":215},[176,800,483],{"class":219},[176,802,803],{"class":185},"\"file\"",[176,805,806],{"class":219},": Path(path).name, ",[176,808,809],{"class":185},"\"rows\"",[176,811,718],{"class":219},[176,813,41],{"class":295},[176,815,311],{"class":219},[176,817,818],{"class":185},"\"revenue\"",[176,820,718],{"class":219},[176,822,823],{"class":295},"0.0",[176,825,604],{"class":219},[176,827,829,832,835,838,840,842,844,846,849,852,855,857,859,861,864,866,868],{"class":61,"line":828},21,[176,830,831],{"class":185},"                \"by_region\"",[176,833,834],{"class":219},": {}, ",[176,836,837],{"class":185},"\"error\"",[176,839,718],{"class":219},[176,841,464],{"class":215},[176,843,467],{"class":185},[176,845,471],{"class":470},[176,847,848],{"class":295},"type",[176,850,851],{"class":219},"(exc).",[176,853,854],{"class":295},"__name__",[176,856,480],{"class":470},[176,858,718],{"class":185},[176,860,471],{"class":470},[176,862,863],{"class":219},"exc",[176,865,480],{"class":470},[176,867,467],{"class":185},[176,869,870],{"class":219},"}\n",[10,872,873],{},"Catching the exception inside the worker is the part that makes a batch job survivable. One corrupt workbook then becomes a row in the report rather than a traceback that takes the other twenty-nine results with it.",[10,875,876],{},"The return value must be picklable, which rules out open file handles, workbook objects and generators — another reason to reduce to numbers and dicts before returning.",[161,878,880],{"id":879},"run-the-pool-and-combine","Run the pool and combine",[166,882,884],{"className":206,"code":883,"language":208,"meta":171,"style":171},"from concurrent.futures import ProcessPoolExecutor, as_completed\nfrom pathlib import Path\n\nimport pandas as pd\n\ndef run(folder=\"monthly\", workers=4):\n    files = sorted(str(p) for p in Path(folder).glob(\"*.xlsx\") if not p.name.startswith(\"~$\"))\n    results, failures = [], []\n\n    with ProcessPoolExecutor(max_workers=workers) as pool:\n        futures = {pool.submit(summarise, f): f for f in files}\n        for future in as_completed(futures):\n            outcome = future.result()\n            (failures if outcome[\"error\"] else results).append(outcome)\n            print(f\"{outcome['file']:24} {'FAILED' if outcome['error'] else outcome['rows']}\")\n\n    combined = {}\n    for outcome in results:\n        for region, revenue in outcome[\"by_region\"].items():\n            combined[region] = combined.get(region, 0.0) + revenue\n\n    summary = pd.DataFrame(sorted(combined.items()), columns=[\"Region\", \"Revenue\"])\n    return summary, failures\n\nif __name__ == \"__main__\":               # required on Windows and macOS spawn\n    summary, failures = run()\n    print(summary)\n    for failure in failures:\n        print(\"failed:\", failure[\"file\"], failure[\"error\"])\n",[173,885,886,897,907,911,921,925,949,991,1001,1005,1023,1043,1056,1066,1086,1143,1147,1157,1170,1187,1207,1211,1243,1251,1256,1276,1287,1295,1308],{"__ignoreMap":171},[176,887,888,890,892,894],{"class":61,"line":178},[176,889,226],{"class":215},[176,891,229],{"class":219},[176,893,216],{"class":215},[176,895,896],{"class":219}," ProcessPoolExecutor, as_completed\n",[176,898,899,901,903,905],{"class":61,"line":223},[176,900,226],{"class":215},[176,902,242],{"class":219},[176,904,216],{"class":215},[176,906,247],{"class":219},[176,908,909],{"class":61,"line":237},[176,910,254],{"emptyLinePlaceholder":253},[176,912,913,915,917,919],{"class":61,"line":250},[176,914,216],{"class":215},[176,916,262],{"class":219},[176,918,265],{"class":215},[176,920,268],{"class":219},[176,922,923],{"class":61,"line":257},[176,924,254],{"emptyLinePlaceholder":253},[176,926,927,929,932,935,937,939,942,944,946],{"class":61,"line":271},[176,928,279],{"class":215},[176,930,931],{"class":282}," run",[176,933,934],{"class":219},"(folder",[176,936,306],{"class":215},[176,938,364],{"class":185},[176,940,941],{"class":219},", workers",[176,943,306],{"class":215},[176,945,431],{"class":295},[176,947,948],{"class":219},"):\n",[176,950,951,954,956,958,960,962,964,966,968,970,973,975,977,980,983,986,989],{"class":61,"line":276},[176,952,953],{"class":219},"    files ",[176,955,306],{"class":215},[176,957,340],{"class":295},[176,959,343],{"class":219},[176,961,346],{"class":295},[176,963,349],{"class":219},[176,965,352],{"class":215},[176,967,355],{"class":219},[176,969,358],{"class":215},[176,971,972],{"class":219}," Path(folder).glob(",[176,974,370],{"class":185},[176,976,434],{"class":219},[176,978,979],{"class":215},"if",[176,981,982],{"class":215}," not",[176,984,985],{"class":219}," p.name.startswith(",[176,987,988],{"class":185},"\"~$\"",[176,990,373],{"class":219},[176,992,993,996,998],{"class":61,"line":289},[176,994,995],{"class":219},"    results, failures ",[176,997,306],{"class":215},[176,999,1000],{"class":219}," [], []\n",[176,1002,1003],{"class":61,"line":327},[176,1004,254],{"emptyLinePlaceholder":253},[176,1006,1007,1009,1012,1014,1016,1019,1021],{"class":61,"line":332},[176,1008,420],{"class":215},[176,1010,1011],{"class":219}," ProcessPoolExecutor(",[176,1013,426],{"class":302},[176,1015,306],{"class":215},[176,1017,1018],{"class":219},"workers) ",[176,1020,265],{"class":215},[176,1022,439],{"class":219},[176,1024,1025,1028,1030,1033,1035,1038,1040],{"class":61,"line":376},[176,1026,1027],{"class":219},"        futures ",[176,1029,306],{"class":215},[176,1031,1032],{"class":219}," {pool.submit(summarise, f): f ",[176,1034,352],{"class":215},[176,1036,1037],{"class":219}," f ",[176,1039,358],{"class":215},[176,1041,1042],{"class":219}," files}\n",[176,1044,1045,1048,1051,1053],{"class":61,"line":381},[176,1046,1047],{"class":215},"        for",[176,1049,1050],{"class":219}," future ",[176,1052,358],{"class":215},[176,1054,1055],{"class":219}," as_completed(futures):\n",[176,1057,1058,1061,1063],{"class":61,"line":406},[176,1059,1060],{"class":219},"            outcome ",[176,1062,306],{"class":215},[176,1064,1065],{"class":219}," future.result()\n",[176,1067,1068,1071,1073,1076,1078,1080,1083],{"class":61,"line":417},[176,1069,1070],{"class":219},"            (failures ",[176,1072,979],{"class":215},[176,1074,1075],{"class":219}," outcome[",[176,1077,837],{"class":185},[176,1079,640],{"class":219},[176,1081,1082],{"class":215},"else",[176,1084,1085],{"class":219}," results).append(outcome)\n",[176,1087,1088,1091,1093,1095,1097,1099,1102,1105,1108,1111,1113,1115,1118,1121,1123,1126,1128,1130,1132,1135,1137,1139,1141],{"class":61,"line":442},[176,1089,1090],{"class":295},"            print",[176,1092,343],{"class":219},[176,1094,464],{"class":215},[176,1096,467],{"class":185},[176,1098,471],{"class":470},[176,1100,1101],{"class":219},"outcome[",[176,1103,1104],{"class":185},"'file'",[176,1106,1107],{"class":219},"]",[176,1109,1110],{"class":215},":24",[176,1112,480],{"class":470},[176,1114,483],{"class":470},[176,1116,1117],{"class":185},"'FAILED'",[176,1119,1120],{"class":215}," if",[176,1122,1075],{"class":219},[176,1124,1125],{"class":185},"'error'",[176,1127,640],{"class":219},[176,1129,1082],{"class":215},[176,1131,1075],{"class":219},[176,1133,1134],{"class":185},"'rows'",[176,1136,1107],{"class":219},[176,1138,480],{"class":470},[176,1140,467],{"class":185},[176,1142,519],{"class":219},[176,1144,1145],{"class":61,"line":456},[176,1146,254],{"emptyLinePlaceholder":253},[176,1148,1149,1152,1154],{"class":61,"line":758},[176,1150,1151],{"class":219},"    combined ",[176,1153,306],{"class":215},[176,1155,1156],{"class":219}," {}\n",[176,1158,1159,1162,1165,1167],{"class":61,"line":771},[176,1160,1161],{"class":215},"    for",[176,1163,1164],{"class":219}," outcome ",[176,1166,358],{"class":215},[176,1168,1169],{"class":219}," results:\n",[176,1171,1172,1174,1177,1179,1181,1184],{"class":61,"line":777},[176,1173,1047],{"class":215},[176,1175,1176],{"class":219}," region, revenue ",[176,1178,358],{"class":215},[176,1180,1075],{"class":219},[176,1182,1183],{"class":185},"\"by_region\"",[176,1185,1186],{"class":219},"].items():\n",[176,1188,1189,1192,1194,1197,1199,1201,1204],{"class":61,"line":796},[176,1190,1191],{"class":219},"            combined[region] ",[176,1193,306],{"class":215},[176,1195,1196],{"class":219}," combined.get(region, ",[176,1198,823],{"class":295},[176,1200,434],{"class":219},[176,1202,1203],{"class":215},"+",[176,1205,1206],{"class":219}," revenue\n",[176,1208,1209],{"class":61,"line":828},[176,1210,254],{"emptyLinePlaceholder":253},[176,1212,1214,1217,1219,1222,1225,1228,1231,1233,1235,1237,1239,1241],{"class":61,"line":1213},22,[176,1215,1216],{"class":219},"    summary ",[176,1218,306],{"class":215},[176,1220,1221],{"class":219}," pd.DataFrame(",[176,1223,1224],{"class":295},"sorted",[176,1226,1227],{"class":219},"(combined.items()), ",[176,1229,1230],{"class":302},"columns",[176,1232,306],{"class":215},[176,1234,319],{"class":219},[176,1236,616],{"class":185},[176,1238,311],{"class":219},[176,1240,637],{"class":185},[176,1242,629],{"class":219},[176,1244,1246,1248],{"class":61,"line":1245},23,[176,1247,292],{"class":215},[176,1249,1250],{"class":219}," summary, failures\n",[176,1252,1254],{"class":61,"line":1253},24,[176,1255,254],{"emptyLinePlaceholder":253},[176,1257,1259,1261,1264,1267,1270,1273],{"class":61,"line":1258},25,[176,1260,979],{"class":215},[176,1262,1263],{"class":295}," __name__",[176,1265,1266],{"class":215}," ==",[176,1268,1269],{"class":185}," \"__main__\"",[176,1271,1272],{"class":219},":               ",[176,1274,1275],{"class":792},"# required on Windows and macOS spawn\n",[176,1277,1279,1282,1284],{"class":61,"line":1278},26,[176,1280,1281],{"class":219},"    summary, failures ",[176,1283,306],{"class":215},[176,1285,1286],{"class":219}," run()\n",[176,1288,1290,1292],{"class":61,"line":1289},27,[176,1291,459],{"class":295},[176,1293,1294],{"class":219},"(summary)\n",[176,1296,1298,1300,1303,1305],{"class":61,"line":1297},28,[176,1299,1161],{"class":215},[176,1301,1302],{"class":219}," failure ",[176,1304,358],{"class":215},[176,1306,1307],{"class":219}," failures:\n",[176,1309,1311,1314,1316,1319,1322,1324,1327,1329],{"class":61,"line":1310},29,[176,1312,1313],{"class":295},"        print",[176,1315,343],{"class":219},[176,1317,1318],{"class":185},"\"failed:\"",[176,1320,1321],{"class":219},", failure[",[176,1323,803],{"class":185},[176,1325,1326],{"class":219},"], failure[",[176,1328,837],{"class":185},[176,1330,629],{"class":219},[10,1332,1333,1334,1337,1338,1341],{},"Two details are easy to skip and painful to debug. The ",[173,1335,1336],{},"if __name__ == \"__main__\""," guard is mandatory wherever processes are spawned rather than forked — without it each worker re-imports the module and starts its own pool. And filtering ",[173,1339,1340],{},"~$"," files matters on any shared drive: those are Excel's lock files, they are not valid workbooks, and they appear whenever a colleague has the folder open.",[161,1343,1345],{"id":1344},"choosing-the-worker-count","Choosing the worker count",[20,1347,29,1352,29,1355,29,1358,29,1361,29,1366,29,1372,29,1377,29,1384,29,1389,29,1393,29,1396,29,1401,29,1407,29,1411,29,1414,29,1419,29,1424,29,1430,29,1434,29,1437,29,1440,29,1445,29,1450,29,1454,29,1457],{"viewBox":1348,"role":23,"ariaLabelledBy":1349,"xmlns":27,"style":28},"0 0 740 216",[1350,1351],"par-workers-t","par-workers-d",[31,1353,1354],{"id":1350},"Worker count against wall time and memory",[35,1356,1357],{"id":1351},"Going from one to four workers roughly quarters the wall time while memory rises in step. Beyond the number of physical cores the time stops improving but memory keeps growing, so the useful setting is the core count capped by the memory budget.",[39,1359],{"x":41,"y":41,"width":42,"height":1360,"fill":44},"216",[54,1362,1365],{"x":56,"y":1363,"style":1364},"28","font-size:12.5px;font-weight:600;fill:var(--muted,#5b6780);text-anchor:middle","each worker parses its own workbook — memory scales with the count",[39,1367],{"x":1368,"y":50,"width":1369,"height":1370,"rx":51,"fill":87,"stroke":1371},"40","150","130","var(--line,#cdd5e6)",[54,1373,1376],{"x":1374,"y":86,"style":1375},"115","font-size:12.5px;font-weight:700;fill:var(--text,#172033);text-anchor:middle","1 worker",[39,1378],{"x":1379,"y":1380,"width":1381,"height":1382,"rx":1383,"fill":52},"70","86","90","60","6",[54,1385,1388],{"x":1374,"y":1386,"style":1387},"122","font-size:12px;font-weight:700;fill:#ffffff;text-anchor:middle","100%",[54,1390,1392],{"x":1374,"y":1391,"style":98},"164","time 60s · 400 MB",[39,1394],{"x":1395,"y":50,"width":1369,"height":1370,"rx":51,"fill":154,"stroke":137,"style":155},"210",[54,1397,1400],{"x":1398,"y":86,"style":1399},"285","font-size:12.5px;font-weight:700;fill:var(--teal-ink,#0b6157);text-anchor:middle","4 workers",[39,1402],{"x":1403,"y":1404,"width":1382,"height":1405,"rx":1383,"fill":1406},"255","116","30","#0f766e",[54,1408,1410],{"x":1398,"y":1409,"style":1387},"137","28%",[54,1412,1413],{"x":1398,"y":1391,"style":98},"time 17s · 1.6 GB",[39,1415],{"x":1416,"y":50,"width":1369,"height":1370,"rx":51,"fill":1417,"stroke":1418},"380","#fdefd8","var(--gold,#b4740a)",[54,1420,1423],{"x":1421,"y":86,"style":1422},"455","font-size:12.5px;font-weight:700;fill:var(--gold-ink,#7a4e06);text-anchor:middle","8 workers",[39,1425],{"x":1426,"y":1427,"width":1382,"height":1428,"rx":1383,"fill":1429},"425","112","34","#8a5808",[54,1431,1433],{"x":1421,"y":1432,"style":1387},"136","26%",[54,1435,1436],{"x":1421,"y":1391,"style":98},"time 16s · 3.2 GB",[39,1438],{"x":1439,"y":50,"width":1369,"height":1370,"rx":51,"fill":123,"stroke":124},"550",[54,1441,1444],{"x":1442,"y":86,"style":1443},"625","font-size:12.5px;font-weight:700;fill:var(--danger,#dc2626);text-anchor:middle","16 workers",[39,1446],{"x":1447,"y":1448,"width":1382,"height":1368,"rx":1383,"fill":1449},"595","106","#d81b73",[54,1451,1453],{"x":1442,"y":1452,"style":1387},"132","41%",[54,1455,1456],{"x":1442,"y":1391,"style":98},"swapping · slower",[54,1458,1460],{"x":56,"y":49,"style":1459},"font-size:11.5px;fill:var(--muted,#5b6780);text-anchor:middle","Illustrative shape: start at the physical core count, then reduce until memory fits.",[10,1462,1463],{},"The illustrated shape is what you should expect rather than a benchmark of your machine: gains flatten at the physical core count, and once total memory exceeds what is available the wall time gets worse, not just flat. Because each worker holds an entire workbook, memory is nearly always the binding constraint with Excel work — four workers on 500 MB files needs two gigabytes before any of your own data structures exist.",[161,1465,1467],{"id":1466},"when-not-to-parallelise","When not to parallelise",[1469,1470,1471,1484],"table",{},[1472,1473,1474],"thead",{},[1475,1476,1477,1481],"tr",{},[1478,1479,1480],"th",{},"Situation",[1478,1482,1483],{},"Better approach",[1485,1486,1487,1496,1504,1512,1520,1528],"tbody",{},[1475,1488,1489,1493],{},[1490,1491,1492],"td",{},"A handful of small files",[1490,1494,1495],{},"A plain loop — pool startup costs more than it saves",[1475,1497,1498,1501],{},[1490,1499,1500],{},"One very large file",[1490,1502,1503],{},"Stream it; parallelism cannot split a single sheet easily",[1475,1505,1506,1509],{},[1490,1507,1508],{},"Downloading the files",[1490,1510,1511],{},"Threads, then process the local copies",[1475,1513,1514,1517],{},[1490,1515,1516],{},"Workers returning big frames",[1490,1518,1519],{},"Aggregate in the worker, or write per-file output to disk",[1475,1521,1522,1525],{},[1490,1523,1524],{},"A memory-capped container",[1490,1526,1527],{},"Fewer workers, or sequential streaming",[1475,1529,1530,1533],{},[1490,1531,1532],{},"Debugging",[1490,1534,1535],{},"Run sequentially first; tracebacks cross process boundaries badly",[10,1537,1538,1539,1542],{},"The last row is a practical habit rather than a rule. Keep the worker function callable on its own and test it on one file directly — ",[173,1540,1541],{},"summarise(\"monthly\u002Fjan.xlsx\")"," — before it goes anywhere near a pool.",[161,1544,1546],{"id":1545},"writing-results-per-file-instead-of-returning-them","Writing results per file instead of returning them",[10,1548,1549],{},"When each file produces substantial output rather than a summary, have the worker write its own file and return only the path. Nothing large crosses the process boundary:",[166,1551,1553],{"className":206,"code":1552,"language":208,"meta":171,"style":171},"from pathlib import Path\n\nimport pandas as pd\n\ndef clean_to_parquet(path, outdir=\"clean\"):\n    Path(outdir).mkdir(exist_ok=True)\n    target = Path(outdir) \u002F (Path(path).stem + \".parquet\")\n    df = pd.read_excel(path, sheet_name=\"Orders\")\n    df[\"Revenue\"] = df[\"Quantity\"] * df[\"Unit_Price\"]\n    df.to_parquet(target, index=False)\n    return {\"file\": Path(path).name, \"rows\": len(df), \"output\": str(target)}\n",[173,1554,1555,1565,1569,1579,1583,1600,1614,1637,1654,1679,1694],{"__ignoreMap":171},[176,1556,1557,1559,1561,1563],{"class":61,"line":178},[176,1558,226],{"class":215},[176,1560,242],{"class":219},[176,1562,216],{"class":215},[176,1564,247],{"class":219},[176,1566,1567],{"class":61,"line":223},[176,1568,254],{"emptyLinePlaceholder":253},[176,1570,1571,1573,1575,1577],{"class":61,"line":237},[176,1572,216],{"class":215},[176,1574,262],{"class":219},[176,1576,265],{"class":215},[176,1578,268],{"class":219},[176,1580,1581],{"class":61,"line":250},[176,1582,254],{"emptyLinePlaceholder":253},[176,1584,1585,1587,1590,1593,1595,1598],{"class":61,"line":257},[176,1586,279],{"class":215},[176,1588,1589],{"class":282}," clean_to_parquet",[176,1591,1592],{"class":219},"(path, outdir",[176,1594,306],{"class":215},[176,1596,1597],{"class":185},"\"clean\"",[176,1599,948],{"class":219},[176,1601,1602,1605,1608,1610,1612],{"class":61,"line":271},[176,1603,1604],{"class":219},"    Path(outdir).mkdir(",[176,1606,1607],{"class":302},"exist_ok",[176,1609,306],{"class":215},[176,1611,681],{"class":295},[176,1613,519],{"class":219},[176,1615,1616,1619,1621,1624,1627,1630,1632,1635],{"class":61,"line":276},[176,1617,1618],{"class":219},"    target ",[176,1620,306],{"class":215},[176,1622,1623],{"class":219}," Path(outdir) ",[176,1625,1626],{"class":215},"\u002F",[176,1628,1629],{"class":219}," (Path(path).stem ",[176,1631,1203],{"class":215},[176,1633,1634],{"class":185}," \".parquet\"",[176,1636,519],{"class":219},[176,1638,1639,1642,1644,1646,1648,1650,1652],{"class":61,"line":289},[176,1640,1641],{"class":219},"    df ",[176,1643,306],{"class":215},[176,1645,594],{"class":219},[176,1647,303],{"class":302},[176,1649,306],{"class":215},[176,1651,601],{"class":185},[176,1653,519],{"class":219},[176,1655,1656,1659,1661,1663,1665,1667,1669,1671,1673,1675,1677],{"class":61,"line":327},[176,1657,1658],{"class":219},"    df[",[176,1660,637],{"class":185},[176,1662,640],{"class":219},[176,1664,306],{"class":215},[176,1666,645],{"class":219},[176,1668,621],{"class":185},[176,1670,640],{"class":219},[176,1672,652],{"class":215},[176,1674,645],{"class":219},[176,1676,626],{"class":185},[176,1678,659],{"class":219},[176,1680,1681,1684,1687,1689,1692],{"class":61,"line":332},[176,1682,1683],{"class":219},"    df.to_parquet(target, ",[176,1685,1686],{"class":302},"index",[176,1688,306],{"class":215},[176,1690,1691],{"class":295},"False",[176,1693,519],{"class":219},[176,1695,1696,1698,1700,1702,1704,1706,1708,1710,1713,1716,1718,1720],{"class":61,"line":376},[176,1697,292],{"class":215},[176,1699,483],{"class":219},[176,1701,803],{"class":185},[176,1703,806],{"class":219},[176,1705,809],{"class":185},[176,1707,718],{"class":219},[176,1709,726],{"class":295},[176,1711,1712],{"class":219},"(df), ",[176,1714,1715],{"class":185},"\"output\"",[176,1717,718],{"class":219},[176,1719,346],{"class":295},[176,1721,1722],{"class":219},"(target)}\n",[10,1724,1725],{},"The parent then reads the Parquet files afterwards — cheaply, and possibly in parallel too. This \"fan out, write, fan in\" shape is what most batch Excel jobs should look like once they outgrow a single loop.",[161,1727,1729],{"id":1728},"threads-for-fetching-processes-for-parsing","Threads for fetching, processes for parsing",[10,1731,1732],{},"Batch jobs over a shared drive or object storage have two phases with opposite bottlenecks, and the fastest arrangement uses both pools:",[20,1734,29,1740,29,1743,29,1746,29,1751,29,1755,29,1761,29,1766,29,1770,29,1774,29,1778,29,1783,29,1787,29,1791,29,1794,29,1797,29,1800],{"viewBox":1735,"role":23,"ariaLabelledBy":1736,"xmlns":27,"style":1739},"14 28 710 175",[1737,1738],"par-two-t","par-two-d","width:100%;max-width:710px;height:auto;display:block;margin:1.5rem auto;font-family:Inter,ui-sans-serif,system-ui,sans-serif",[31,1741,1742],{"id":1737},"A thread pool downloads, a process pool parses",[35,1744,1745],{"id":1738},"Downloading thirty workbooks is input-output bound, so a thread pool overlaps the waiting. Parsing them is processor bound, so a process pool uses every core. Each phase uses the pool that matches its bottleneck.",[39,1747],{"x":1748,"y":1363,"width":1749,"height":1750,"fill":44},"14","710","175",[39,1752],{"x":1405,"y":1753,"width":1754,"height":1427,"rx":1748,"fill":1417,"stroke":1418,"style":155},"44","300",[54,1756,1760],{"x":1757,"y":1758,"style":1759},"180","74","font-size:13px;font-weight:700;fill:var(--gold-ink,#7a4e06);text-anchor:middle","phase 1 · fetch",[54,1762,1765],{"x":1757,"y":1763,"style":1764},"102","font-size:11.5px;fill:var(--text,#172033);text-anchor:middle","ThreadPoolExecutor",[54,1767,1769],{"x":1757,"y":1768,"style":1459},"124","waiting on the network,",[54,1771,1773],{"x":1757,"y":1772,"style":1459},"142","not on the CPU",[61,1775],{"x1":1776,"y1":84,"x2":1777,"y2":84,"stroke":67,"style":155},"336","386",[1779,1780],"polygon",{"points":1781,"fill":1782},"390,100 378,94 378,106","#5b6780",[39,1784],{"x":1785,"y":1753,"width":1786,"height":1427,"rx":1748,"fill":154,"stroke":137,"style":155},"392","316",[54,1788,1790],{"x":1439,"y":1758,"style":1789},"font-size:13px;font-weight:700;fill:var(--teal-ink,#0b6157);text-anchor:middle","phase 2 · parse",[54,1792,1793],{"x":1439,"y":1763,"style":1764},"ProcessPoolExecutor",[54,1795,1796],{"x":1439,"y":1768,"style":1459},"XML parsing holds the lock,",[54,1798,1799],{"x":1439,"y":1772,"style":1459},"so it needs real processes",[54,1801,1803],{"x":56,"y":1802,"style":1459},"184","Local copies land on disk between the phases — cheap, and restartable.",[10,1805,1806],{},"Writing the downloads to a local folder between the phases is worth the disk it uses. A batch that fails halfway can be restarted against the files already fetched, and the parse phase becomes reproducible without touching the network at all.",[161,1808,1810],{"id":1809},"common-pitfalls-and-gotchas","Common pitfalls and gotchas",[1469,1812,1813,1826],{},[1472,1814,1815],{},[1475,1816,1817,1820,1823],{},[1478,1818,1819],{},"Symptom",[1478,1821,1822],{},"Cause",[1478,1824,1825],{},"Fix",[1485,1827,1828,1841,1854,1868,1879,1892,1909],{},[1475,1829,1830,1833,1836],{},[1490,1831,1832],{},"No speed-up at all",[1490,1834,1835],{},"Used threads for CPU-bound parsing",[1490,1837,1838,1839],{},"Use ",[173,1840,1793],{},[1475,1842,1843,1846,1851],{},[1490,1844,1845],{},"Workers spawn recursively",[1490,1847,1848,1849],{},"Missing ",[173,1850,1336],{},[1490,1852,1853],{},"Add the guard around the pool",[1475,1855,1856,1862,1865],{},[1490,1857,1858,1861],{},[173,1859,1860],{},"MemoryError"," with several workers",[1490,1863,1864],{},"Each worker holds a full workbook",[1490,1866,1867],{},"Fewer workers, or stream inside the worker",[1475,1869,1870,1873,1876],{},[1490,1871,1872],{},"Whole batch dies on one file",[1490,1874,1875],{},"Exception raised out of a worker",[1490,1877,1878],{},"Catch inside the worker and return the error",[1475,1880,1881,1886,1889],{},[1490,1882,1883],{},[173,1884,1885],{},"PicklingError",[1490,1887,1888],{},"Returned an unpicklable object",[1490,1890,1891],{},"Return dicts, lists and numbers",[1475,1893,1894,1897,1904],{},[1490,1895,1896],{},"Weird files in the batch",[1490,1898,1899,1900,1903],{},"Excel lock files (",[173,1901,1902],{},"~$name.xlsx",")",[1490,1905,1906,1907],{},"Filter names starting with ",[173,1908,1340],{},[1475,1910,1911,1914,1920],{},[1490,1912,1913],{},"Results in the wrong order",[1490,1915,1916,1919],{},[173,1917,1918],{},"as_completed"," yields by completion",[1490,1921,1922,1923],{},"Sort afterwards, or use ",[173,1924,1925],{},"pool.map",[161,1927,1929],{"id":1928},"keep-a-progress-signal","Keep a progress signal",[10,1931,1932,1933,1935],{},"A batch that prints nothing for four minutes is indistinguishable from a batch that has hung. ",[173,1934,1918],{}," gives you a natural place to report progress, and the count is free:",[166,1937,1939],{"className":206,"code":1938,"language":208,"meta":171,"style":171},"from concurrent.futures import ProcessPoolExecutor, as_completed\n\ndef run_with_progress(files, workers=4):\n    done = 0\n    results = []\n    with ProcessPoolExecutor(max_workers=workers) as pool:\n        futures = [pool.submit(summarise, f) for f in files]\n        for future in as_completed(futures):\n            outcome = future.result()\n            results.append(outcome)\n            done += 1\n            status = \"error\" if outcome[\"error\"] else f\"{outcome['rows']:,} rows\"\n            print(f\"[{done}\u002F{len(files)}] {outcome['file']:24} {status}\", flush=True)\n    return results\n",[173,1940,1941,1951,1955,1971,1981,1991,2007,2025,2035,2043,2048,2059,2099,2162],{"__ignoreMap":171},[176,1942,1943,1945,1947,1949],{"class":61,"line":178},[176,1944,226],{"class":215},[176,1946,229],{"class":219},[176,1948,216],{"class":215},[176,1950,896],{"class":219},[176,1952,1953],{"class":61,"line":223},[176,1954,254],{"emptyLinePlaceholder":253},[176,1956,1957,1959,1962,1965,1967,1969],{"class":61,"line":237},[176,1958,279],{"class":215},[176,1960,1961],{"class":282}," run_with_progress",[176,1963,1964],{"class":219},"(files, workers",[176,1966,306],{"class":215},[176,1968,431],{"class":295},[176,1970,948],{"class":219},[176,1972,1973,1976,1978],{"class":61,"line":250},[176,1974,1975],{"class":219},"    done ",[176,1977,306],{"class":215},[176,1979,1980],{"class":295}," 0\n",[176,1982,1983,1986,1988],{"class":61,"line":257},[176,1984,1985],{"class":219},"    results ",[176,1987,306],{"class":215},[176,1989,1990],{"class":219}," []\n",[176,1992,1993,1995,1997,1999,2001,2003,2005],{"class":61,"line":271},[176,1994,420],{"class":215},[176,1996,1011],{"class":219},[176,1998,426],{"class":302},[176,2000,306],{"class":215},[176,2002,1018],{"class":219},[176,2004,265],{"class":215},[176,2006,439],{"class":219},[176,2008,2009,2011,2013,2016,2018,2020,2022],{"class":61,"line":276},[176,2010,1027],{"class":219},[176,2012,306],{"class":215},[176,2014,2015],{"class":219}," [pool.submit(summarise, f) ",[176,2017,352],{"class":215},[176,2019,1037],{"class":219},[176,2021,358],{"class":215},[176,2023,2024],{"class":219}," files]\n",[176,2026,2027,2029,2031,2033],{"class":61,"line":289},[176,2028,1047],{"class":215},[176,2030,1050],{"class":219},[176,2032,358],{"class":215},[176,2034,1055],{"class":219},[176,2036,2037,2039,2041],{"class":61,"line":327},[176,2038,1060],{"class":219},[176,2040,306],{"class":215},[176,2042,1065],{"class":219},[176,2044,2045],{"class":61,"line":332},[176,2046,2047],{"class":219},"            results.append(outcome)\n",[176,2049,2050,2053,2056],{"class":61,"line":376},[176,2051,2052],{"class":219},"            done ",[176,2054,2055],{"class":215},"+=",[176,2057,2058],{"class":295}," 1\n",[176,2060,2061,2064,2066,2069,2071,2073,2075,2077,2079,2082,2084,2086,2088,2090,2092,2094,2096],{"class":61,"line":381},[176,2062,2063],{"class":219},"            status ",[176,2065,306],{"class":215},[176,2067,2068],{"class":185}," \"error\"",[176,2070,1120],{"class":215},[176,2072,1075],{"class":219},[176,2074,837],{"class":185},[176,2076,640],{"class":219},[176,2078,1082],{"class":215},[176,2080,2081],{"class":215}," f",[176,2083,467],{"class":185},[176,2085,471],{"class":470},[176,2087,1101],{"class":219},[176,2089,1134],{"class":185},[176,2091,1107],{"class":219},[176,2093,511],{"class":215},[176,2095,480],{"class":470},[176,2097,2098],{"class":185}," rows\"\n",[176,2100,2101,2103,2105,2107,2110,2112,2115,2117,2119,2121,2123,2126,2128,2130,2132,2134,2136,2138,2140,2142,2144,2147,2149,2151,2153,2156,2158,2160],{"class":61,"line":406},[176,2102,1090],{"class":295},[176,2104,343],{"class":219},[176,2106,464],{"class":215},[176,2108,2109],{"class":185},"\"[",[176,2111,471],{"class":470},[176,2113,2114],{"class":219},"done",[176,2116,480],{"class":470},[176,2118,1626],{"class":185},[176,2120,471],{"class":470},[176,2122,726],{"class":295},[176,2124,2125],{"class":219},"(files)",[176,2127,480],{"class":470},[176,2129,640],{"class":185},[176,2131,471],{"class":470},[176,2133,1101],{"class":219},[176,2135,1104],{"class":185},[176,2137,1107],{"class":219},[176,2139,1110],{"class":215},[176,2141,480],{"class":470},[176,2143,483],{"class":470},[176,2145,2146],{"class":219},"status",[176,2148,480],{"class":470},[176,2150,467],{"class":185},[176,2152,311],{"class":219},[176,2154,2155],{"class":302},"flush",[176,2157,306],{"class":215},[176,2159,681],{"class":295},[176,2161,519],{"class":219},[176,2163,2164,2166],{"class":61,"line":417},[176,2165,292],{"class":215},[176,2167,2168],{"class":219}," results\n",[10,2170,2171,2174,2175,2179],{},[173,2172,2173],{},"flush=True"," matters when the job runs under a scheduler that buffers output — without it the whole log arrives at the end, which defeats the point. The same counter is what you would emit to a metrics system or write into the job's log file for the ",[14,2176,2178],{"href":2177},"\u002Fautomating-reporting-workflows\u002Ferror-handling-and-logging-in-excel-automation\u002F","error-handling layer"," to alert on.",[161,2181,2183],{"id":2182},"performance-and-scale-notes","Performance and scale notes",[10,2185,2186],{},"Expect a speed-up close to the physical core count for parsing-heavy work, and nothing beyond it — hyper-threads add little because the work is memory-bandwidth bound as much as it is arithmetic. Pool startup is tens of milliseconds per worker, so batches under a second of total work are better off sequential. If the same folder is processed repeatedly, the biggest win is not more cores but caching: convert each workbook to Parquet once, then let subsequent runs read that, which is usually faster than any parallel Excel parse.",[161,2188,2190],{"id":2189},"conclusion","Conclusion",[10,2192,2193],{},"A folder of workbooks parallelises almost perfectly, provided each worker parses one file, catches its own exceptions, and returns something small. Use processes rather than threads for parsing, guard the entry point, filter Excel's lock files, and set the worker count from your memory budget rather than your core count. When per-file output is large, write it from the worker and return the path.",[161,2195,2197],{"id":2196},"frequently-asked-questions","Frequently asked questions",[10,2199,2200,2204,2205,2207],{},[2201,2202,2203],"strong",{},"Why processes rather than threads?","\nParsing ",[173,2206,202],{}," is CPU-bound Python work, so it holds the global interpreter lock. Threads take turns on one core; processes genuinely run at the same time.",[10,2209,2210,2213],{},[2201,2211,2212],{},"How many workers should I use?","\nStart at the number of physical cores, then reduce until peak memory fits — each worker holds its own workbook, so memory is the usual constraint rather than CPU.",[10,2215,2216,2219],{},[2201,2217,2218],{},"Can workers return DataFrames?","\nThey can, but every result is pickled and copied back to the parent. Return aggregates where possible and only return frames when they are small.",[10,2221,2222,2225],{},[2201,2223,2224],{},"What happens if one file is corrupt?","\nCatch the exception inside the worker and return it as data. An uncaught exception surfaces when you read that future's result, and it is easy to lose the rest of the batch to it.",[161,2227,2229],{"id":2228},"related","Related",[10,2231,2232],{},"Up to the parent guide:",[2234,2235,2236],"ul",{},[2237,2238,2239,2241],"li",{},[14,2240,17],{"href":16}," — the other levers to try before adding cores.",[10,2243,2244],{},"Related guides:",[2234,2246,2247,2254,2261,2267],{},[2237,2248,2249,2253],{},[14,2250,2252],{"href":2251},"\u002Fgetting-started-with-python-excel-automation\u002Fworking-with-multiple-excel-sheets-in-python\u002Fcombine-multiple-excel-files-into-one-python\u002F","Combine Multiple Excel Files into One with Python"," — the sequential version of the same job.",[2237,2255,2256,2260],{},[14,2257,2259],{"href":2258},"\u002Fadvanced-data-transformation-and-cleaning\u002Fworking-with-large-excel-files-in-python\u002Fconvert-excel-to-csv-with-python\u002F","Convert Excel to CSV with Python"," — the per-file output format that makes fan-in cheap.",[2237,2262,2263,2266],{},[14,2264,2265],{"href":2177},"Error Handling and Logging in Excel Automation"," — reporting the failures a batch collects.",[2237,2268,2269,2273],{},[14,2270,2272],{"href":2271},"\u002Fautomating-reporting-workflows\u002Fscheduling-python-excel-scripts-with-cron\u002Fschedule-recurring-excel-reports-with-apscheduler\u002F","Schedule Recurring Excel Reports with APScheduler"," — running the batch unattended.",[2275,2276,2277],"style",{},"html pre.shiki code .sMTad, html code.shiki .sMTad{--shiki-default:#6F42C1;--shiki-dark:#FFB757}html pre.shiki code .srMev, html code.shiki .srMev{--shiki-default:#032F62;--shiki-dark:#ADDCFF}html .default .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .dark .shiki span {color: var(--shiki-dark);background: var(--shiki-dark-bg);font-style: var(--shiki-dark-font-style);font-weight: var(--shiki-dark-font-weight);text-decoration: var(--shiki-dark-text-decoration);}html.dark .shiki span {color: var(--shiki-dark);background: var(--shiki-dark-bg);font-style: var(--shiki-dark-font-style);font-weight: var(--shiki-dark-font-weight);text-decoration: var(--shiki-dark-text-decoration);}html pre.shiki code .s-kum, html code.shiki .s-kum{--shiki-default:#D73A49;--shiki-dark:#FF9492}html pre.shiki code .skGVy, html code.shiki .skGVy{--shiki-default:#24292E;--shiki-dark:#F0F3F6}html pre.shiki code .s_Opv, html code.shiki .s_Opv{--shiki-default:#6F42C1;--shiki-dark:#DBB7FF}html pre.shiki code .sP0c6, html code.shiki .sP0c6{--shiki-default:#005CC5;--shiki-dark:#91CBFF}html pre.shiki code .sa561, html code.shiki .sa561{--shiki-default:#E36209;--shiki-dark:#FFB757}html pre.shiki code .sSjpA, html code.shiki .sSjpA{--shiki-default:#005CC5;--shiki-dark:#FF9492}html pre.shiki code .s-wDw, html code.shiki .s-wDw{--shiki-default:#6A737D;--shiki-dark:#BDC4CC}",{"title":171,"searchDepth":223,"depth":223,"links":2279},[2280,2281,2282,2283,2284,2285,2286,2287,2288,2289,2290,2291,2292,2293],{"id":163,"depth":223,"text":164},{"id":195,"depth":223,"text":196},{"id":525,"depth":223,"text":526},{"id":879,"depth":223,"text":880},{"id":1344,"depth":223,"text":1345},{"id":1466,"depth":223,"text":1467},{"id":1545,"depth":223,"text":1546},{"id":1728,"depth":223,"text":1729},{"id":1809,"depth":223,"text":1810},{"id":1928,"depth":223,"text":1929},{"id":2182,"depth":223,"text":2183},{"id":2189,"depth":223,"text":2190},{"id":2196,"depth":223,"text":2197},{"id":2228,"depth":223,"text":2229},"2026-08-01","Use every core on a folder of workbooks with ProcessPoolExecutor — why threads do not help, how to return small results, error isolation per file, and when parallelism makes things worse.","md",[2298,2300,2302,2304],{"q":2203,"a":2299},"Parsing xlsx is CPU-bound Python work, so it holds the global interpreter lock. Threads take turns on one core; processes genuinely run at the same time.",{"q":2212,"a":2301},"Start at the number of physical cores, then reduce until peak memory fits — each worker holds its own workbook, so memory is the usual constraint rather than CPU.",{"q":2218,"a":2303},"They can, but every result is pickled and copied back to the parent. Return aggregates where possible and only return frames when they are small.",{"q":2224,"a":2305},"Catch the exception inside the worker and return it as data. An uncaught exception surfaces when you read that future's result, and it is easy to lose the rest of the batch to it.",{"breadcrumb":2307},[2308,2310,2313,2314],{"name":2309,"item":1626},"Home",{"name":2311,"item":2312},"Advanced Data Transformation and Cleaning","\u002Fadvanced-data-transformation-and-cleaning\u002F",{"name":17,"item":16},{"name":5,"item":2315},"\u002Fadvanced-data-transformation-and-cleaning\u002Fworking-with-large-excel-files-in-python\u002Fprocess-multiple-excel-files-in-parallel-with-python\u002F","\u002Fadvanced-data-transformation-and-cleaning\u002Fworking-with-large-excel-files-in-python\u002Fprocess-multiple-excel-files-in-parallel-with-python",{"title":5,"description":2318},"Parallelise Excel processing with concurrent.futures: processes not threads, per-file worker functions, small return values, error handling per file, and choosing the worker count.","process-multiple-excel-files-in-parallel-with-python","advanced-data-transformation-and-cleaning\u002Fworking-with-large-excel-files-in-python\u002Fprocess-multiple-excel-files-in-parallel-with-python\u002Findex","how-to","4XEr1dIFnwuK_lHy39g9FitnnE03MtFtQ4_FAqXzdXc",[2324,2327],{"title":2259,"path":2325,"stem":2326,"children":-1},"\u002Fadvanced-data-transformation-and-cleaning\u002Fworking-with-large-excel-files-in-python\u002Fconvert-excel-to-csv-with-python","advanced-data-transformation-and-cleaning\u002Fworking-with-large-excel-files-in-python\u002Fconvert-excel-to-csv-with-python\u002Findex",{"title":2328,"path":2329,"stem":2330,"children":-1},"Read a Large Excel File in Chunks with pandas","\u002Fadvanced-data-transformation-and-cleaning\u002Fworking-with-large-excel-files-in-python\u002Fread-large-excel-file-in-chunks-with-pandas","advanced-data-transformation-and-cleaning\u002Fworking-with-large-excel-files-in-python\u002Fread-large-excel-file-in-chunks-with-pandas\u002Findex",1785584462728]