[{"data":1,"prerenderedAt":2235},["ShallowReactive",2],{"doc:\u002Fadvanced-data-transformation-and-cleaning\u002Fmoving-data-between-excel-and-databases":3,"surround:\u002Fadvanced-data-transformation-and-cleaning\u002Fmoving-data-between-excel-and-databases":2227},{"id":4,"title":5,"body":6,"dateModified":2197,"datePublished":2197,"description":2198,"extension":2199,"faq":2200,"meta":2212,"navigation":201,"path":2219,"seo":2220,"slug":2223,"stem":2224,"type":2225,"__hash__":2226},"docs\u002Fadvanced-data-transformation-and-cleaning\u002Fmoving-data-between-excel-and-databases\u002Findex.md","Moving Data Between Excel and Databases",{"type":7,"value":8,"toc":2183},"minimark",[9,13,22,159,164,167,260,267,270,274,286,559,582,586,593,723,983,999,1003,1006,1129,1148,1151,1155,1162,1256,1259,1266,1340,1343,1347,1350,1461,1469,1473,1476,1479,1716,1719,1722,1817,1820,1824,1827,1924,1938,1942,1950,1983,1986,1994,1998,2058,2062,2071,2077,2097,2111,2117,2121,2179],[10,11,12],"p",{},"Most reporting sits between two worlds. The numbers live in a database — a warehouse, a Postgres instance behind an application, a SQL Server the finance system writes to — and the people who need them work in Excel. The job is not to choose between the two but to move data across the boundary reliably, in both directions, without losing types, duplicating rows or hard-coding a password into a script.",[10,14,15,16,21],{},"This topic within ",[17,18,20],"a",{"href":19},"\u002Fadvanced-data-transformation-and-cleaning\u002F","Advanced Data Transformation and Cleaning"," covers that boundary: connecting from Python, exporting query results into a formatted workbook, loading a spreadsheet back into a table without damaging what is already there, keeping dates and identifiers intact through the round trip, and running the whole thing on a schedule.",[23,24,32,33,32,37,32,41,32,48,32,58,32,65,32,70,32,75,32,80,32,84,32,89,32,94,32,97,32,102,32,106,32,112,32,115,32,119,32,124,32,128,32,132,32,137,32,140,32,143,32,146,32,151,32,155],"svg",{"viewBox":25,"role":26,"ariaLabelledBy":27,"xmlns":30,"style":31},"0 0 760 244","img",[28,29],"db-loop-t","db-loop-d","http:\u002F\u002Fwww.w3.org\u002F2000\u002Fsvg","width:100%;max-width:760px;height:auto;display:block;margin:1.5rem auto;font-family:Inter,ui-sans-serif,system-ui,sans-serif","\n  ",[34,35,36],"title",{"id":28},"The round trip between a database and a workbook",[38,39,40],"desc",{"id":29},"Python reads from the database with SQLAlchemy and writes a formatted workbook for readers. In the other direction it reads a spreadsheet people have filled in, validates it, loads it into a staging table and only then merges it into the live table.",[42,43],"rect",{"x":44,"y":44,"width":45,"height":46,"fill":47},"0","760","244","#ffffff",[42,49],{"x":50,"y":51,"width":52,"height":53,"rx":54,"fill":55,"stroke":56,"style":57},"20","82","164","84","14","#ebebfd","var(--brand,#5b5cf0)","stroke-width:2px",[59,60,64],"text",{"x":61,"y":62,"style":63},"102","116","font-size:13px;font-weight:700;fill:var(--brand-strong,#4338ca);text-anchor:middle","database",[59,66,69],{"x":61,"y":67,"style":68},"140","font-size:11.5px;fill:var(--muted,#5b6780);text-anchor:middle","the source of truth",[42,71],{"x":72,"y":51,"width":52,"height":53,"rx":54,"fill":73,"stroke":74,"style":57},"298","#5b5cf0","var(--brand-strong,#4338ca)",[59,76,79],{"x":77,"y":62,"style":78},"380","font-size:13px;font-weight:700;fill:#ffffff;text-anchor:middle","Python",[59,81,83],{"x":77,"y":67,"style":82},"font-size:11.5px;fill:rgba(255,255,255,0.9);text-anchor:middle","SQLAlchemy + pandas",[42,85],{"x":86,"y":51,"width":52,"height":53,"rx":54,"fill":87,"stroke":88,"style":57},"576","#d9f4f1","var(--teal,#0f9488)",[59,90,93],{"x":91,"y":62,"style":92},"658","font-size:13px;font-weight:700;fill:var(--teal-ink,#0b6157);text-anchor:middle","workbook",[59,95,96],{"x":91,"y":67,"style":68},"the interface",[98,99],"path",{"d":100,"stroke":56,"style":57,"fill":101},"M 188 106 L 294 106","none",[103,104],"polygon",{"points":105,"fill":73},"298,106 288,101 288,111",[59,107,111],{"x":108,"y":109,"style":110},"241","96","font-size:11px;font-weight:600;fill:var(--brand-strong,#4338ca);text-anchor:middle","read_sql",[98,113],{"d":114,"stroke":88,"style":57,"fill":101},"M 466 106 L 572 106",[103,116],{"points":117,"fill":118},"576,106 566,101 566,111","#0f766e",[59,120,123],{"x":121,"y":109,"style":122},"519","font-size:11px;font-weight:600;fill:var(--teal-ink,#0b6157);text-anchor:middle","to_excel",[98,125],{"d":126,"stroke":127,"style":57,"fill":101},"M 572 146 L 466 146","var(--gold,#b4740a)",[103,129],{"points":130,"fill":131},"462,146 472,141 472,151","#8a5808",[59,133,136],{"x":121,"y":134,"style":135},"166","font-size:11px;font-weight:600;fill:var(--gold-ink,#7a4e06);text-anchor:middle","read_excel",[98,138],{"d":139,"stroke":127,"style":57,"fill":101},"M 294 146 L 188 146",[103,141],{"points":142,"fill":131},"184,146 194,141 194,151",[59,144,145],{"x":108,"y":134,"style":135},"to_sql",[59,147,150],{"x":77,"y":148,"style":149},"42","font-size:13px;font-weight:600;fill:var(--muted,#5b6780);text-anchor:middle","Out to readers, back from contributors",[59,152,154],{"x":77,"y":153,"style":68},"212","The return leg is the risky one: a spreadsheet arrives with edited types, blank rows and",[59,156,158],{"x":77,"y":157,"style":68},"230","duplicated keys, so it is staged and validated before anything touches the live table.",[160,161,163],"h2",{"id":162},"connect-once-in-one-place","Connect once, in one place",[10,165,166],{},"Every database Python talks to is reachable through a SQLAlchemy URL, which means the only thing that changes between SQLite, Postgres, MySQL and SQL Server is a string:",[168,169,174],"pre",{"className":170,"code":171,"language":172,"meta":173,"style":173},"language-python shiki shiki-themes github-light github-dark-high-contrast","from sqlalchemy import create_engine\n\n# sqlite:\u002F\u002F\u002Freports.db\n# postgresql+psycopg:\u002F\u002Fuser:pw@host:5432\u002Fsales\n# mysql+pymysql:\u002F\u002Fuser:pw@host:3306\u002Fsales\n# mssql+pyodbc:\u002F\u002Fuser:pw@host\u002Fsales?driver=ODBC+Driver+18+for+SQL+Server\nengine = create_engine(os.environ[\"REPORT_DB_URL\"], pool_pre_ping=True)\n","python","",[175,176,177,196,203,210,216,222,228],"code",{"__ignoreMap":173},[178,179,182,186,190,193],"span",{"class":180,"line":181},"line",1,[178,183,185],{"class":184},"s-kum","from",[178,187,189],{"class":188},"skGVy"," sqlalchemy ",[178,191,192],{"class":184},"import",[178,194,195],{"class":188}," create_engine\n",[178,197,199],{"class":180,"line":198},2,[178,200,202],{"emptyLinePlaceholder":201},true,"\n",[178,204,206],{"class":180,"line":205},3,[178,207,209],{"class":208},"s-wDw","# sqlite:\u002F\u002F\u002Freports.db\n",[178,211,213],{"class":180,"line":212},4,[178,214,215],{"class":208},"# postgresql+psycopg:\u002F\u002Fuser:pw@host:5432\u002Fsales\n",[178,217,219],{"class":180,"line":218},5,[178,220,221],{"class":208},"# mysql+pymysql:\u002F\u002Fuser:pw@host:3306\u002Fsales\n",[178,223,225],{"class":180,"line":224},6,[178,226,227],{"class":208},"# mssql+pyodbc:\u002F\u002Fuser:pw@host\u002Fsales?driver=ODBC+Driver+18+for+SQL+Server\n",[178,229,231,234,237,240,244,247,251,253,257],{"class":180,"line":230},7,[178,232,233],{"class":188},"engine ",[178,235,236],{"class":184},"=",[178,238,239],{"class":188}," create_engine(os.environ[",[178,241,243],{"class":242},"srMev","\"REPORT_DB_URL\"",[178,245,246],{"class":188},"], ",[178,248,250],{"class":249},"sa561","pool_pre_ping",[178,252,236],{"class":184},[178,254,256],{"class":255},"sP0c6","True",[178,258,259],{"class":188},")\n",[10,261,262,263,266],{},"Two habits are worth adopting at this line rather than later. Read the URL from the environment, never from the source or a committed config file — it contains a password, and a database URL in version control is the most common credential leak in reporting code. And set ",[175,264,265],{},"pool_pre_ping=True",", because a scheduled job holds its engine across a long idle period and a stale connection otherwise surfaces as a confusing error on the first query of the morning.",[10,268,269],{},"The engine is created once per process. Creating one inside a loop opens a new pool per iteration and exhausts the server's connection limit long before the report finishes.",[160,271,273],{"id":272},"export-a-query-into-a-workbook-people-will-read","Export a query into a workbook people will read",[10,275,276,277,280,281,285],{},"The naive export — ",[175,278,279],{},"pd.read_sql(...).to_excel(...)"," — produces a correct, unreadable grid. The version worth scheduling adds the formatting that makes the file usable, which is where this topic meets ",[17,282,284],{"href":283},"\u002Fformatting-and-charting-excel-reports-with-python\u002F","Formatting and Charting Excel Reports with Python",":",[168,287,289],{"className":170,"code":288,"language":172,"meta":173,"style":173},"import pandas as pd\nfrom sqlalchemy import text\n\nQUERY = text(\"\"\"\n    SELECT region, order_date, SUM(amount) AS amount\n    FROM orders\n    WHERE order_date >= :start AND order_date \u003C :end\n    GROUP BY region, order_date\n    ORDER BY region, order_date\n\"\"\")\n\nwith engine.connect() as conn:\n    df = pd.read_sql(QUERY, conn, params={\"start\": \"2026-07-01\", \"end\": \"2026-08-01\"})\n\nwith pd.ExcelWriter(\"regional.xlsx\", engine=\"openpyxl\",\n                    datetime_format=\"yyyy-mm-dd\") as writer:\n    df.to_excel(writer, sheet_name=\"Detail\", index=False)\n    (df.groupby(\"region\", as_index=False)[\"amount\"].sum()\n       .to_excel(writer, sheet_name=\"Summary\", index=False))\n",[175,290,291,304,315,319,333,338,343,348,354,360,368,373,387,434,439,463,482,508,535],{"__ignoreMap":173},[178,292,293,295,298,301],{"class":180,"line":181},[178,294,192],{"class":184},[178,296,297],{"class":188}," pandas ",[178,299,300],{"class":184},"as",[178,302,303],{"class":188}," pd\n",[178,305,306,308,310,312],{"class":180,"line":198},[178,307,185],{"class":184},[178,309,189],{"class":188},[178,311,192],{"class":184},[178,313,314],{"class":188}," text\n",[178,316,317],{"class":180,"line":205},[178,318,202],{"emptyLinePlaceholder":201},[178,320,321,324,327,330],{"class":180,"line":212},[178,322,323],{"class":255},"QUERY",[178,325,326],{"class":184}," =",[178,328,329],{"class":188}," text(",[178,331,332],{"class":242},"\"\"\"\n",[178,334,335],{"class":180,"line":218},[178,336,337],{"class":242},"    SELECT region, order_date, SUM(amount) AS amount\n",[178,339,340],{"class":180,"line":224},[178,341,342],{"class":242},"    FROM orders\n",[178,344,345],{"class":180,"line":230},[178,346,347],{"class":242},"    WHERE order_date >= :start AND order_date \u003C :end\n",[178,349,351],{"class":180,"line":350},8,[178,352,353],{"class":242},"    GROUP BY region, order_date\n",[178,355,357],{"class":180,"line":356},9,[178,358,359],{"class":242},"    ORDER BY region, order_date\n",[178,361,363,366],{"class":180,"line":362},10,[178,364,365],{"class":242},"\"\"\"",[178,367,259],{"class":188},[178,369,371],{"class":180,"line":370},11,[178,372,202],{"emptyLinePlaceholder":201},[178,374,376,379,382,384],{"class":180,"line":375},12,[178,377,378],{"class":184},"with",[178,380,381],{"class":188}," engine.connect() ",[178,383,300],{"class":184},[178,385,386],{"class":188}," conn:\n",[178,388,390,393,395,398,400,403,406,408,411,414,417,420,423,426,428,431],{"class":180,"line":389},13,[178,391,392],{"class":188},"    df ",[178,394,236],{"class":184},[178,396,397],{"class":188}," pd.read_sql(",[178,399,323],{"class":255},[178,401,402],{"class":188},", conn, ",[178,404,405],{"class":249},"params",[178,407,236],{"class":184},[178,409,410],{"class":188},"{",[178,412,413],{"class":242},"\"start\"",[178,415,416],{"class":188},": ",[178,418,419],{"class":242},"\"2026-07-01\"",[178,421,422],{"class":188},", ",[178,424,425],{"class":242},"\"end\"",[178,427,416],{"class":188},[178,429,430],{"class":242},"\"2026-08-01\"",[178,432,433],{"class":188},"})\n",[178,435,437],{"class":180,"line":436},14,[178,438,202],{"emptyLinePlaceholder":201},[178,440,442,444,447,450,452,455,457,460],{"class":180,"line":441},15,[178,443,378],{"class":184},[178,445,446],{"class":188}," pd.ExcelWriter(",[178,448,449],{"class":242},"\"regional.xlsx\"",[178,451,422],{"class":188},[178,453,454],{"class":249},"engine",[178,456,236],{"class":184},[178,458,459],{"class":242},"\"openpyxl\"",[178,461,462],{"class":188},",\n",[178,464,466,469,471,474,477,479],{"class":180,"line":465},16,[178,467,468],{"class":249},"                    datetime_format",[178,470,236],{"class":184},[178,472,473],{"class":242},"\"yyyy-mm-dd\"",[178,475,476],{"class":188},") ",[178,478,300],{"class":184},[178,480,481],{"class":188}," writer:\n",[178,483,485,488,491,493,496,498,501,503,506],{"class":180,"line":484},17,[178,486,487],{"class":188},"    df.to_excel(writer, ",[178,489,490],{"class":249},"sheet_name",[178,492,236],{"class":184},[178,494,495],{"class":242},"\"Detail\"",[178,497,422],{"class":188},[178,499,500],{"class":249},"index",[178,502,236],{"class":184},[178,504,505],{"class":255},"False",[178,507,259],{"class":188},[178,509,511,514,517,519,522,524,526,529,532],{"class":180,"line":510},18,[178,512,513],{"class":188},"    (df.groupby(",[178,515,516],{"class":242},"\"region\"",[178,518,422],{"class":188},[178,520,521],{"class":249},"as_index",[178,523,236],{"class":184},[178,525,505],{"class":255},[178,527,528],{"class":188},")[",[178,530,531],{"class":242},"\"amount\"",[178,533,534],{"class":188},"].sum()\n",[178,536,538,541,543,545,548,550,552,554,556],{"class":180,"line":537},19,[178,539,540],{"class":188},"       .to_excel(writer, ",[178,542,490],{"class":249},[178,544,236],{"class":184},[178,546,547],{"class":242},"\"Summary\"",[178,549,422],{"class":188},[178,551,500],{"class":249},[178,553,236],{"class":184},[178,555,505],{"class":255},[178,557,558],{"class":188},"))\n",[10,560,561,562,565,566,569,570,572,573,576,577,581],{},"The ",[175,563,564],{},":start"," and ",[175,567,568],{},":end"," placeholders with a ",[175,571,405],{}," dictionary are not stylistic. A query assembled with an f-string from a value that came out of a spreadsheet cell is an injection waiting to happen, and bound parameters also let the database cache the plan. ",[175,574,575],{},"datetime_format"," on the writer saves a formatting pass: without it, dates land as serial numbers or as text depending on the engine. ",[17,578,580],{"href":579},"\u002Fadvanced-data-transformation-and-cleaning\u002Fmoving-data-between-excel-and-databases\u002Fexport-sql-query-results-to-excel-with-python\u002F","Export SQL Query Results to Excel with Python"," covers multi-sheet exports, column widths and chunked reads for large results.",[160,583,585],{"id":584},"stage-the-return-leg","Stage the return leg",[10,587,588,589,592],{},"Loading a spreadsheet into a database is the direction that goes wrong, because a workbook filled in by people contains everything a database schema does not expect: a blank row someone left at the bottom, a total row, an ID typed as text, a date in two formats, the same key twice. Writing that straight into a live table with ",[175,590,591],{},"if_exists=\"append\""," publishes all of it.",[23,594,32,600,32,603,32,606,32,610,32,618,32,624,32,629,32,633,32,638,32,641,32,644,32,649,32,652,32,656,32,661,32,664,32,670,32,674,32,678,32,681,32,685,32,688,32,694,32,697,32,704,32,710,32,715,32,719],{"viewBox":595,"role":26,"ariaLabelledBy":596,"xmlns":30,"style":599},"0 0 740 252",[597,598],"db-stage-t","db-stage-d","width:100%;max-width:740px;height:auto;display:block;margin:1.5rem auto;font-family:Inter,ui-sans-serif,system-ui,sans-serif",[34,601,602],{"id":597},"Loading a spreadsheet through a staging table",[38,604,605],{"id":598},"The workbook is read into a DataFrame and written to a staging table. Validation runs against the staging table — row counts, required columns, duplicate keys, type checks. Only if every check passes does a single transaction merge the rows into the live table; otherwise the staging table is dropped and the live table is untouched.",[42,607],{"x":44,"y":44,"width":608,"height":609,"fill":47},"740","252",[42,611],{"x":612,"y":613,"width":67,"height":614,"rx":615,"fill":616,"stroke":617,"style":57},"16","90","66","12","#f0f2f5","var(--line,#cdd5e6)",[59,619,623],{"x":620,"y":621,"style":622},"86","118","font-size:12px;font-weight:700;fill:var(--text,#172033);text-anchor:middle","upload.xlsx",[59,625,628],{"x":620,"y":626,"style":627},"139","font-size:11px;fill:var(--muted,#5b6780);text-anchor:middle","filled in by people",[42,630],{"x":631,"y":613,"width":67,"height":614,"rx":615,"fill":632,"stroke":127,"style":57},"196","#fdefd8",[59,634,637],{"x":635,"y":621,"style":636},"266","font-size:12px;font-weight:700;fill:var(--gold-ink,#7a4e06);text-anchor:middle","staging table",[59,639,640],{"x":635,"y":626,"style":627},"everything, as typed",[42,642],{"x":643,"y":613,"width":67,"height":614,"rx":615,"fill":55,"stroke":56,"style":57},"376",[59,645,648],{"x":646,"y":621,"style":647},"446","font-size:12px;font-weight:700;fill:var(--brand-strong,#4338ca);text-anchor:middle","checks",[59,650,651],{"x":646,"y":626,"style":627},"keys · types · counts",[42,653],{"x":654,"y":613,"width":655,"height":614,"rx":615,"fill":87,"stroke":88,"style":57},"556","168",[59,657,660],{"x":658,"y":621,"style":659},"640","font-size:12px;font-weight:700;fill:var(--teal-ink,#0b6157);text-anchor:middle","live table",[59,662,663],{"x":658,"y":626,"style":627},"one transaction",[180,665],{"x1":666,"y1":667,"x2":668,"y2":667,"stroke":669,"style":57},"160","123","190","var(--muted,#5b6780)",[103,671],{"points":672,"fill":673},"194,123 184,118 184,128","#5b6780",[180,675],{"x1":676,"y1":667,"x2":677,"y2":667,"stroke":669,"style":57},"340","370",[103,679],{"points":680,"fill":673},"374,123 364,118 364,128",[180,682],{"x1":683,"y1":667,"x2":684,"y2":667,"stroke":88,"style":57},"520","550",[103,686],{"points":687,"fill":118},"554,123 544,118 544,128",[59,689,693],{"x":690,"y":691,"style":692},"537","80","font-size:10.5px;font-weight:600;fill:var(--teal-ink,#0b6157);text-anchor:middle","pass",[59,695,696],{"x":677,"y":148,"style":149},"Nothing reaches the live table until every check has passed",[42,698],{"x":631,"y":699,"width":700,"height":701,"rx":615,"fill":702,"stroke":703,"style":57},"184","320","52","#fce9e9","var(--accent-ink,#be185d)",[59,705,709],{"x":706,"y":707,"style":708},"356","206","font-size:11.5px;font-weight:700;fill:var(--accent-ink,#be185d);text-anchor:middle","on failure: drop staging, report which rows",[59,711,714],{"x":706,"y":712,"style":713},"226","font-size:11px;fill:var(--text,#172033);text-anchor:middle","the live table never saw the bad file",[180,716],{"x1":646,"y1":666,"x2":717,"y2":718,"stroke":703,"style":57},"400","180",[103,720],{"points":721,"fill":722},"396,182 407,176 405,186","#d81b73",[168,724,726],{"className":170,"code":725,"language":172,"meta":173,"style":173},"def load_via_staging(df, engine, table=\"sales_actuals\"):\n    staging = f\"{table}__staging\"\n    df.to_sql(staging, engine, if_exists=\"replace\", index=False, chunksize=5_000)\n\n    with engine.begin() as conn:                 # one transaction, commits or rolls back\n        dupes = conn.execute(text(\n            f\"SELECT COUNT(*) FROM (SELECT order_id FROM {staging} \"\n            f\"GROUP BY order_id HAVING COUNT(*) > 1) d\")).scalar()\n        if dupes:\n            raise ValueError(f\"{dupes} duplicate order_id value(s) in the upload\")\n\n        conn.execute(text(f\"DELETE FROM {table} WHERE period = :p\"), {\"p\": \"2026-07\"})\n        conn.execute(text(f\"INSERT INTO {table} SELECT * FROM {staging}\"))\n        conn.execute(text(f\"DROP TABLE {staging}\"))\n",[175,727,728,748,773,806,810,826,836,854,864,872,900,904,936,964],{"__ignoreMap":173},[178,729,730,733,737,740,742,745],{"class":180,"line":181},[178,731,732],{"class":184},"def",[178,734,736],{"class":735},"s_Opv"," load_via_staging",[178,738,739],{"class":188},"(df, engine, table",[178,741,236],{"class":184},[178,743,744],{"class":242},"\"sales_actuals\"",[178,746,747],{"class":188},"):\n",[178,749,750,753,755,758,761,764,767,770],{"class":180,"line":198},[178,751,752],{"class":188},"    staging ",[178,754,236],{"class":184},[178,756,757],{"class":184}," f",[178,759,760],{"class":242},"\"",[178,762,410],{"class":763},"sSjpA",[178,765,766],{"class":188},"table",[178,768,769],{"class":763},"}",[178,771,772],{"class":242},"__staging\"\n",[178,774,775,778,781,783,786,788,790,792,794,796,799,801,804],{"class":180,"line":205},[178,776,777],{"class":188},"    df.to_sql(staging, engine, ",[178,779,780],{"class":249},"if_exists",[178,782,236],{"class":184},[178,784,785],{"class":242},"\"replace\"",[178,787,422],{"class":188},[178,789,500],{"class":249},[178,791,236],{"class":184},[178,793,505],{"class":255},[178,795,422],{"class":188},[178,797,798],{"class":249},"chunksize",[178,800,236],{"class":184},[178,802,803],{"class":255},"5_000",[178,805,259],{"class":188},[178,807,808],{"class":180,"line":212},[178,809,202],{"emptyLinePlaceholder":201},[178,811,812,815,818,820,823],{"class":180,"line":218},[178,813,814],{"class":184},"    with",[178,816,817],{"class":188}," engine.begin() ",[178,819,300],{"class":184},[178,821,822],{"class":188}," conn:                 ",[178,824,825],{"class":208},"# one transaction, commits or rolls back\n",[178,827,828,831,833],{"class":180,"line":224},[178,829,830],{"class":188},"        dupes ",[178,832,236],{"class":184},[178,834,835],{"class":188}," conn.execute(text(\n",[178,837,838,841,844,846,849,851],{"class":180,"line":230},[178,839,840],{"class":184},"            f",[178,842,843],{"class":242},"\"SELECT COUNT(*) FROM (SELECT order_id FROM ",[178,845,410],{"class":763},[178,847,848],{"class":188},"staging",[178,850,769],{"class":763},[178,852,853],{"class":242}," \"\n",[178,855,856,858,861],{"class":180,"line":350},[178,857,840],{"class":184},[178,859,860],{"class":242},"\"GROUP BY order_id HAVING COUNT(*) > 1) d\"",[178,862,863],{"class":188},")).scalar()\n",[178,865,866,869],{"class":180,"line":356},[178,867,868],{"class":184},"        if",[178,870,871],{"class":188}," dupes:\n",[178,873,874,877,880,883,886,888,890,893,895,898],{"class":180,"line":362},[178,875,876],{"class":184},"            raise",[178,878,879],{"class":255}," ValueError",[178,881,882],{"class":188},"(",[178,884,885],{"class":184},"f",[178,887,760],{"class":242},[178,889,410],{"class":763},[178,891,892],{"class":188},"dupes",[178,894,769],{"class":763},[178,896,897],{"class":242}," duplicate order_id value(s) in the upload\"",[178,899,259],{"class":188},[178,901,902],{"class":180,"line":370},[178,903,202],{"emptyLinePlaceholder":201},[178,905,906,909,911,914,916,918,920,923,926,929,931,934],{"class":180,"line":375},[178,907,908],{"class":188},"        conn.execute(text(",[178,910,885],{"class":184},[178,912,913],{"class":242},"\"DELETE FROM ",[178,915,410],{"class":763},[178,917,766],{"class":188},[178,919,769],{"class":763},[178,921,922],{"class":242}," WHERE period = :p\"",[178,924,925],{"class":188},"), {",[178,927,928],{"class":242},"\"p\"",[178,930,416],{"class":188},[178,932,933],{"class":242},"\"2026-07\"",[178,935,433],{"class":188},[178,937,938,940,942,945,947,949,951,954,956,958,960,962],{"class":180,"line":389},[178,939,908],{"class":188},[178,941,885],{"class":184},[178,943,944],{"class":242},"\"INSERT INTO ",[178,946,410],{"class":763},[178,948,766],{"class":188},[178,950,769],{"class":763},[178,952,953],{"class":242}," SELECT * FROM ",[178,955,410],{"class":763},[178,957,848],{"class":188},[178,959,769],{"class":763},[178,961,760],{"class":242},[178,963,558],{"class":188},[178,965,966,968,970,973,975,977,979,981],{"class":180,"line":436},[178,967,908],{"class":188},[178,969,885],{"class":184},[178,971,972],{"class":242},"\"DROP TABLE ",[178,974,410],{"class":763},[178,976,848],{"class":188},[178,978,769],{"class":763},[178,980,760],{"class":242},[178,982,558],{"class":188},[10,984,985,988,989,993,994,998],{},[175,986,987],{},"engine.begin()"," is the important call: it opens a transaction that commits on a clean exit and rolls back on any exception, so a failure halfway through the delete-and-insert leaves the live table exactly as it was. The delete-then-insert pattern also makes the load idempotent — re-running the same file for the same period produces the same table rather than a doubled one, which matters as soon as anything ",[17,990,992],{"href":991},"\u002Fautomating-reporting-workflows\u002Ferror-handling-and-logging-in-excel-automation\u002Fretry-a-failed-excel-report-job-in-python\u002F","retries",". ",[17,995,997],{"href":996},"\u002Fadvanced-data-transformation-and-cleaning\u002Fmoving-data-between-excel-and-databases\u002Fload-an-excel-file-into-a-sql-database-with-pandas\u002F","Load an Excel File into a SQL Database with pandas"," works through the type mapping and the validation in detail.",[160,1000,1002],{"id":1001},"decide-where-the-work-happens","Decide where the work happens",[10,1004,1005],{},"The single biggest performance decision at this boundary is not which library you use but which side does the aggregation. A database can group ten million rows without sending any of them anywhere; pandas can only group rows it has already received, which means transferring them across the network, materialising them in memory, and then discarding almost all of them.",[23,1007,32,1012,32,1015,32,1018,32,1020,32,1025,32,1031,32,1035,32,1039,32,1043,32,1046,32,1052,32,1055,32,1060,32,1064,32,1069,32,1073,32,1077,32,1081,32,1084,32,1087,32,1089,32,1093,32,1096,32,1101,32,1103,32,1108,32,1110,32,1113,32,1116,32,1119,32,1124],{"viewBox":1008,"role":26,"ariaLabelledBy":1009,"xmlns":30,"style":599},"0 0 740 244",[1010,1011],"db-where-t","db-where-d",[34,1013,1014],{"id":1010},"Aggregating in SQL versus aggregating in pandas",[38,1016,1017],{"id":1011},"Grouping in the database transfers only the summary rows, so the report holds a few hundred rows in memory. Selecting everything and grouping in pandas transfers every detail row across the network first, then discards almost all of them — the same answer for far more time and memory.",[42,1019],{"x":44,"y":44,"width":608,"height":46,"fill":47},[59,1021,1024],{"x":699,"y":1022,"style":1023},"32","font-size:12.5px;font-weight:700;fill:var(--teal-ink,#0b6157);text-anchor:middle","GROUP BY in the database",[42,1026],{"x":1027,"y":701,"width":1028,"height":1029,"rx":1030,"fill":87,"stroke":88,"style":57},"24","120","46","8",[59,1032,1034],{"x":53,"y":1033,"style":713},"72","10,000,000",[59,1036,1038],{"x":53,"y":1037,"style":627},"89","rows in the table",[180,1040],{"x1":1041,"y1":1042,"x2":631,"y2":1042,"stroke":88,"style":57},"150","75",[103,1044],{"points":1045,"fill":118},"200,75 190,70 190,80",[59,1047,1051],{"x":1048,"y":1049,"style":1050},"173","64","font-size:10px;font-weight:600;fill:var(--teal-ink,#0b6157);text-anchor:middle","group",[42,1053],{"x":1054,"y":701,"width":67,"height":1029,"rx":1030,"fill":73,"stroke":74,"style":57},"204",[59,1056,1059],{"x":1057,"y":1033,"style":1058},"274","font-size:11px;font-weight:700;fill:#ffffff;text-anchor:middle","220 summary rows",[59,1061,1063],{"x":1057,"y":1037,"style":1062},"font-size:11px;fill:rgba(255,255,255,0.9);text-anchor:middle","cross the network",[59,1065,1068],{"x":699,"y":1066,"style":1067},"130","font-size:11.5px;fill:var(--text,#172033);text-anchor:middle","seconds, and a few megabytes",[59,1070,1072],{"x":699,"y":1071,"style":627},"154","indexes on the filter and group columns",[59,1074,1076],{"x":699,"y":1075,"style":627},"172","do the work you would otherwise pay for twice",[59,1078,1080],{"x":654,"y":1022,"style":1079},"font-size:12.5px;font-weight:700;fill:var(--accent-ink,#be185d);text-anchor:middle","SELECT * then groupby in pandas",[42,1082],{"x":1083,"y":701,"width":1028,"height":1029,"rx":1030,"fill":702,"stroke":703,"style":57},"396",[59,1085,1034],{"x":1086,"y":1033,"style":713},"456",[59,1088,1038],{"x":1086,"y":1037,"style":627},[180,1090],{"x1":1091,"y1":1042,"x2":1092,"y2":1042,"stroke":703,"style":57},"522","568",[103,1094],{"points":1095,"fill":722},"572,75 562,70 562,80",[59,1097,1100],{"x":1098,"y":1049,"style":1099},"545","font-size:10px;font-weight:600;fill:var(--accent-ink,#be185d);text-anchor:middle","all of it",[42,1102],{"x":86,"y":701,"width":67,"height":1029,"rx":1030,"fill":702,"stroke":703,"style":57},[59,1104,1107],{"x":1105,"y":1033,"style":1106},"646","font-size:11px;font-weight:700;fill:var(--accent-ink,#be185d);text-anchor:middle","10,000,000 rows",[59,1109,1063],{"x":1105,"y":1037,"style":713},[59,1111,1112],{"x":654,"y":1066,"style":1067},"minutes, and gigabytes of memory",[59,1114,1115],{"x":654,"y":1071,"style":627},"to produce the identical 220 rows —",[59,1117,1118],{"x":654,"y":1075,"style":627},"and to fail on the day the table doubles",[42,1120],{"x":1028,"y":1121,"width":1122,"height":148,"rx":1123,"fill":632,"stroke":127},"192","500","10",[59,1125,1128],{"x":677,"y":1126,"style":1127},"218","font-size:11.5px;font-weight:700;fill:var(--gold-ink,#7a4e06);text-anchor:middle","Push filters and grouping down; keep shaping and formatting in Python",[10,1130,1131,1132,1136,1137,422,1140,1143,1144,1147],{},"The useful division is simple. Anything that ",[1133,1134,1135],"em",{},"reduces"," rows — filters, joins against reference tables, ",[175,1138,1139],{},"GROUP BY",[175,1141,1142],{},"DISTINCT"," — belongs in SQL, where indexes exist and nothing has to be transferred. Anything that ",[1133,1145,1146],{},"reshapes"," what is left — pivoting for presentation, deriving display columns, formatting, ordering the sheets — belongs in pandas, where the code is easier to read and the result is going into a workbook anyway.",[10,1149,1150],{},"The exception is worth naming, because it is the reason people end up on the wrong side: a query that is awkward to express in SQL, run once a month over a small table, is not worth an hour of window-function debugging. The rule is about magnitude, not principle. When the transfer is a few thousand rows, do whatever is clearest; when it is millions, the database is not optional.",[160,1152,1154],{"id":1153},"give-the-report-its-own-credentials","Give the report its own credentials",[10,1156,1157,1158,1161],{},"A reporting job needs to read a handful of tables and, at most, write to one. It does not need the application's account. Creating a dedicated read-only role costs five minutes and removes an entire category of accident — a mistyped statement in a script that only has ",[175,1159,1160],{},"SELECT"," cannot delete anything:",[168,1163,1167],{"className":1164,"code":1165,"language":1166,"meta":173,"style":173},"language-sql shiki shiki-themes github-light github-dark-high-contrast","CREATE ROLE report_reader LOGIN PASSWORD 'set-from-a-secret-store';\nGRANT CONNECT ON DATABASE sales TO report_reader;\nGRANT USAGE ON SCHEMA public TO report_reader;\nGRANT SELECT ON orders, customers, products TO report_reader;\n-- and nothing else: no INSERT, no UPDATE, no DDL\n","sql",[175,1168,1169,1192,1215,1235,1251],{"__ignoreMap":173},[178,1170,1171,1174,1177,1180,1183,1186,1189],{"class":180,"line":181},[178,1172,1173],{"class":184},"CREATE",[178,1175,1176],{"class":184}," ROLE",[178,1178,1179],{"class":188}," report_reader ",[178,1181,1182],{"class":184},"LOGIN",[178,1184,1185],{"class":184}," PASSWORD",[178,1187,1188],{"class":242}," 'set-from-a-secret-store'",[178,1190,1191],{"class":188},";\n",[178,1193,1194,1197,1200,1203,1206,1209,1212],{"class":180,"line":198},[178,1195,1196],{"class":184},"GRANT",[178,1198,1199],{"class":184}," CONNECT",[178,1201,1202],{"class":184}," ON",[178,1204,1205],{"class":184}," DATABASE",[178,1207,1208],{"class":188}," sales ",[178,1210,1211],{"class":184},"TO",[178,1213,1214],{"class":188}," report_reader;\n",[178,1216,1217,1219,1222,1225,1228,1231,1233],{"class":180,"line":205},[178,1218,1196],{"class":184},[178,1220,1221],{"class":188}," USAGE ",[178,1223,1224],{"class":184},"ON",[178,1226,1227],{"class":184}," SCHEMA",[178,1229,1230],{"class":188}," public ",[178,1232,1211],{"class":184},[178,1234,1214],{"class":188},[178,1236,1237,1239,1242,1244,1247,1249],{"class":180,"line":212},[178,1238,1196],{"class":184},[178,1240,1241],{"class":184}," SELECT",[178,1243,1202],{"class":184},[178,1245,1246],{"class":188}," orders, customers, products ",[178,1248,1211],{"class":184},[178,1250,1214],{"class":188},[178,1252,1253],{"class":180,"line":218},[178,1254,1255],{"class":208},"-- and nothing else: no INSERT, no UPDATE, no DDL\n",[10,1257,1258],{},"Where the job also loads data back, give it a second role with write access to its own staging schema only, and keep the two connection strings separate in the environment. That way the export half of the pipeline physically cannot modify anything, and the load half is scoped to the tables it owns.",[10,1260,1261,1262,1265],{},"Two operational habits go with this. Set a statement timeout so a runaway report query cannot hold locks or saturate the server for an hour — most databases accept it per session, and a reporting connection is exactly the place for it. And name the connection, through the application name in the URL or a ",[175,1263,1264],{},"SET application_name",", so that when a DBA sees a heavy query at 06:00 they can tell which report it belongs to instead of guessing.",[168,1267,1269],{"className":170,"code":1268,"language":172,"meta":173,"style":173},"engine = create_engine(\n    os.environ[\"REPORT_DB_URL\"],\n    pool_pre_ping=True,\n    connect_args={\"application_name\": \"monthly-regional-report\",\n                  \"options\": \"-c statement_timeout=120000\"},   # 2 minutes\n)\n",[175,1270,1271,1280,1290,1301,1320,1336],{"__ignoreMap":173},[178,1272,1273,1275,1277],{"class":180,"line":181},[178,1274,233],{"class":188},[178,1276,236],{"class":184},[178,1278,1279],{"class":188}," create_engine(\n",[178,1281,1282,1285,1287],{"class":180,"line":198},[178,1283,1284],{"class":188},"    os.environ[",[178,1286,243],{"class":242},[178,1288,1289],{"class":188},"],\n",[178,1291,1292,1295,1297,1299],{"class":180,"line":205},[178,1293,1294],{"class":249},"    pool_pre_ping",[178,1296,236],{"class":184},[178,1298,256],{"class":255},[178,1300,462],{"class":188},[178,1302,1303,1306,1308,1310,1313,1315,1318],{"class":180,"line":212},[178,1304,1305],{"class":249},"    connect_args",[178,1307,236],{"class":184},[178,1309,410],{"class":188},[178,1311,1312],{"class":242},"\"application_name\"",[178,1314,416],{"class":188},[178,1316,1317],{"class":242},"\"monthly-regional-report\"",[178,1319,462],{"class":188},[178,1321,1322,1325,1327,1330,1333],{"class":180,"line":218},[178,1323,1324],{"class":242},"                  \"options\"",[178,1326,416],{"class":188},[178,1328,1329],{"class":242},"\"-c statement_timeout=120000\"",[178,1331,1332],{"class":188},"},   ",[178,1334,1335],{"class":208},"# 2 minutes\n",[178,1337,1338],{"class":180,"line":224},[178,1339,259],{"class":188},[10,1341,1342],{},"Both settings turn a report that can quietly become an incident into one that fails on its own terms, in a way that names itself in the server's logs.",[160,1344,1346],{"id":1345},"protect-the-types-across-the-boundary","Protect the types across the boundary",[10,1348,1349],{},"Excel and SQL disagree about types in ways that bite quietly rather than loudly. Four cases account for nearly all of it:",[766,1351,1352,1368],{},[1353,1354,1355],"thead",{},[1356,1357,1358,1362,1365],"tr",{},[1359,1360,1361],"th",{},"What it is",[1359,1363,1364],{},"What goes wrong",[1359,1366,1367],{},"What to do",[1369,1370,1371,1390,1419,1441],"tbody",{},[1356,1372,1373,1380,1383],{},[1374,1375,1376,1377],"td",{},"An identifier like ",[175,1378,1379],{},"00123",[1374,1381,1382],{},"Excel drops the leading zeros; pandas reads it as an integer",[1374,1384,1385,1386,1389],{},"Read with ",[175,1387,1388],{},"dtype={\"order_id\": str}"," and store as text",[1356,1391,1392,1395,1409],{},[1374,1393,1394],{},"An ID column with one blank",[1374,1396,1397,1398,1401,1402,1405,1406],{},"Becomes ",[175,1399,1400],{},"float64",", so ",[175,1403,1404],{},"1"," displays as ",[175,1407,1408],{},"1.0",[1374,1410,1411,1412,1415,1416],{},"Use pandas' nullable ",[175,1413,1414],{},"Int64",", or read as ",[175,1417,1418],{},"str",[1356,1420,1421,1424,1431],{},[1374,1422,1423],{},"A date",[1374,1425,1426,1427,1430],{},"Arrives as a serial number, a string or a ",[175,1428,1429],{},"Timestamp"," depending on the path",[1374,1432,1433,1434,1437,1438],{},"Coerce with ",[175,1435,1436],{},"pd.to_datetime(..., errors=\"coerce\")"," and check for ",[175,1439,1440],{},"NaT",[1356,1442,1443,1446,1452],{},[1374,1444,1445],{},"A currency amount",[1374,1447,1448,1449],{},"Rounding differences between float and ",[175,1450,1451],{},"NUMERIC",[1374,1453,1454,1455,1457,1458],{},"Round explicitly before writing; store as ",[175,1456,1451],{},", not ",[175,1459,1460],{},"FLOAT",[10,1462,1463,1464,1468],{},"The fix is always to be explicit at the read, because a wrong type detected at the write is already a corrupted DataFrame. This is the same discipline that ",[17,1465,1467],{"href":1466},"\u002Fadvanced-data-transformation-and-cleaning\u002Fvalidating-excel-data-with-python\u002Fcheck-excel-data-types-with-pandas\u002F","Check Excel Data Types with pandas"," applies to spreadsheets that never touch a database.",[160,1470,1472],{"id":1471},"reconcile-before-anyone-else-does","Reconcile before anyone else does",[10,1474,1475],{},"The failure that damages a reporting pipeline most is not a crash — it is a workbook that arrives on time, opens cleanly, and disagrees with the system it came from. By the time somebody notices, the number has usually been quoted in a meeting.",[10,1477,1478],{},"Two cheap checks catch nearly all of it, and both belong in the job rather than in a spreadsheet somebody maintains on the side. The first is a count-and-sum reconciliation against the source, run immediately after the extract:",[168,1480,1482],{"className":170,"code":1481,"language":172,"meta":173,"style":173},"CONTROL = text(\"\"\"\n    SELECT COUNT(*) AS rows, COALESCE(SUM(amount), 0) AS total\n    FROM orders WHERE order_date >= :start AND order_date \u003C :end\n\"\"\")\n\nwith engine.connect() as conn:\n    control = conn.execute(CONTROL, params).mappings().one()\n\nif len(df) != control[\"rows\"]:\n    raise ValueError(f\"extract has {len(df):,} rows, source has {control['rows']:,}\")\nif abs(df[\"amount\"].sum() - float(control[\"total\"])) > 0.01:\n    raise ValueError(f\"extract totals {df['amount'].sum():,.2f}, \"\n                     f\"source totals {float(control['total']):,.2f}\")\n",[175,1483,1484,1495,1500,1505,1511,1515,1525,1540,1544,1567,1616,1655,1687],{"__ignoreMap":173},[178,1485,1486,1489,1491,1493],{"class":180,"line":181},[178,1487,1488],{"class":255},"CONTROL",[178,1490,326],{"class":184},[178,1492,329],{"class":188},[178,1494,332],{"class":242},[178,1496,1497],{"class":180,"line":198},[178,1498,1499],{"class":242},"    SELECT COUNT(*) AS rows, COALESCE(SUM(amount), 0) AS total\n",[178,1501,1502],{"class":180,"line":205},[178,1503,1504],{"class":242},"    FROM orders WHERE order_date >= :start AND order_date \u003C :end\n",[178,1506,1507,1509],{"class":180,"line":212},[178,1508,365],{"class":242},[178,1510,259],{"class":188},[178,1512,1513],{"class":180,"line":218},[178,1514,202],{"emptyLinePlaceholder":201},[178,1516,1517,1519,1521,1523],{"class":180,"line":224},[178,1518,378],{"class":184},[178,1520,381],{"class":188},[178,1522,300],{"class":184},[178,1524,386],{"class":188},[178,1526,1527,1530,1532,1535,1537],{"class":180,"line":230},[178,1528,1529],{"class":188},"    control ",[178,1531,236],{"class":184},[178,1533,1534],{"class":188}," conn.execute(",[178,1536,1488],{"class":255},[178,1538,1539],{"class":188},", params).mappings().one()\n",[178,1541,1542],{"class":180,"line":350},[178,1543,202],{"emptyLinePlaceholder":201},[178,1545,1546,1549,1552,1555,1558,1561,1564],{"class":180,"line":356},[178,1547,1548],{"class":184},"if",[178,1550,1551],{"class":255}," len",[178,1553,1554],{"class":188},"(df) ",[178,1556,1557],{"class":184},"!=",[178,1559,1560],{"class":188}," control[",[178,1562,1563],{"class":242},"\"rows\"",[178,1565,1566],{"class":188},"]:\n",[178,1568,1569,1572,1574,1576,1578,1581,1583,1586,1589,1592,1594,1597,1599,1602,1605,1608,1610,1612,1614],{"class":180,"line":362},[178,1570,1571],{"class":184},"    raise",[178,1573,879],{"class":255},[178,1575,882],{"class":188},[178,1577,885],{"class":184},[178,1579,1580],{"class":242},"\"extract has ",[178,1582,410],{"class":763},[178,1584,1585],{"class":255},"len",[178,1587,1588],{"class":188},"(df)",[178,1590,1591],{"class":184},":,",[178,1593,769],{"class":763},[178,1595,1596],{"class":242}," rows, source has ",[178,1598,410],{"class":763},[178,1600,1601],{"class":188},"control[",[178,1603,1604],{"class":242},"'rows'",[178,1606,1607],{"class":188},"]",[178,1609,1591],{"class":184},[178,1611,769],{"class":763},[178,1613,760],{"class":242},[178,1615,259],{"class":188},[178,1617,1618,1620,1623,1626,1628,1631,1634,1637,1640,1643,1646,1649,1652],{"class":180,"line":370},[178,1619,1548],{"class":184},[178,1621,1622],{"class":255}," abs",[178,1624,1625],{"class":188},"(df[",[178,1627,531],{"class":242},[178,1629,1630],{"class":188},"].sum() ",[178,1632,1633],{"class":184},"-",[178,1635,1636],{"class":255}," float",[178,1638,1639],{"class":188},"(control[",[178,1641,1642],{"class":242},"\"total\"",[178,1644,1645],{"class":188},"])) ",[178,1647,1648],{"class":184},">",[178,1650,1651],{"class":255}," 0.01",[178,1653,1654],{"class":188},":\n",[178,1656,1657,1659,1661,1663,1665,1668,1670,1673,1676,1679,1682,1684],{"class":180,"line":375},[178,1658,1571],{"class":184},[178,1660,879],{"class":255},[178,1662,882],{"class":188},[178,1664,885],{"class":184},[178,1666,1667],{"class":242},"\"extract totals ",[178,1669,410],{"class":763},[178,1671,1672],{"class":188},"df[",[178,1674,1675],{"class":242},"'amount'",[178,1677,1678],{"class":188},"].sum()",[178,1680,1681],{"class":184},":,.2f",[178,1683,769],{"class":763},[178,1685,1686],{"class":242},", \"\n",[178,1688,1689,1692,1695,1697,1700,1702,1705,1708,1710,1712,1714],{"class":180,"line":389},[178,1690,1691],{"class":184},"                     f",[178,1693,1694],{"class":242},"\"source totals ",[178,1696,410],{"class":763},[178,1698,1699],{"class":255},"float",[178,1701,1639],{"class":188},[178,1703,1704],{"class":242},"'total'",[178,1706,1707],{"class":188},"])",[178,1709,1681],{"class":184},[178,1711,769],{"class":763},[178,1713,760],{"class":242},[178,1715,259],{"class":188},[10,1717,1718],{},"Running the control query in the same connection and the same transaction as the extract is what makes it meaningful — against a busy table, a second connection can legitimately see different rows, and a check that fails at random gets switched off within a fortnight.",[10,1720,1721],{},"The second is a comparison against the previous run. A month that is 3% up on the last one is unremarkable; one that is 60% down usually means a filter changed, a join lost rows, or the source loaded late. Keeping the control totals from each run in a small table or JSON file makes that check a subtraction:",[168,1723,1725],{"className":170,"code":1724,"language":172,"meta":173,"style":173},"previous = read_state().get(\"total\")\nif previous and abs(df[\"amount\"].sum() - previous) \u002F previous > 0.4:\n    log.warning(\"total moved %.0f%% against the previous run — check the source\",\n                (df[\"amount\"].sum() - previous) \u002F previous * 100)\n",[175,1726,1727,1741,1776,1792],{"__ignoreMap":173},[178,1728,1729,1732,1734,1737,1739],{"class":180,"line":181},[178,1730,1731],{"class":188},"previous ",[178,1733,236],{"class":184},[178,1735,1736],{"class":188}," read_state().get(",[178,1738,1642],{"class":242},[178,1740,259],{"class":188},[178,1742,1743,1745,1748,1751,1753,1755,1757,1759,1761,1764,1767,1769,1771,1774],{"class":180,"line":198},[178,1744,1548],{"class":184},[178,1746,1747],{"class":188}," previous ",[178,1749,1750],{"class":184},"and",[178,1752,1622],{"class":255},[178,1754,1625],{"class":188},[178,1756,531],{"class":242},[178,1758,1630],{"class":188},[178,1760,1633],{"class":184},[178,1762,1763],{"class":188}," previous) ",[178,1765,1766],{"class":184},"\u002F",[178,1768,1747],{"class":188},[178,1770,1648],{"class":184},[178,1772,1773],{"class":255}," 0.4",[178,1775,1654],{"class":188},[178,1777,1778,1781,1784,1787,1790],{"class":180,"line":205},[178,1779,1780],{"class":188},"    log.warning(",[178,1782,1783],{"class":242},"\"total moved ",[178,1785,1786],{"class":763},"%.0f%%",[178,1788,1789],{"class":242}," against the previous run — check the source\"",[178,1791,462],{"class":188},[178,1793,1794,1797,1799,1801,1803,1805,1807,1809,1812,1815],{"class":180,"line":212},[178,1795,1796],{"class":188},"                (df[",[178,1798,531],{"class":242},[178,1800,1630],{"class":188},[178,1802,1633],{"class":184},[178,1804,1763],{"class":188},[178,1806,1766],{"class":184},[178,1808,1747],{"class":188},[178,1810,1811],{"class":184},"*",[178,1813,1814],{"class":255}," 100",[178,1816,259],{"class":188},[10,1818,1819],{},"Whether a large movement should stop the job or merely warn depends on the report. For a month-end pack that goes to the board, stopping and asking a human is right. For a daily operational extract where genuine swings happen, a warning in the log and a note on the summary sheet is enough. What is never right is having no opinion at all, because then the first person to see an implausible number is the person least equipped to explain it.",[160,1821,1823],{"id":1822},"pull-in-what-has-no-database-at-all","Pull in what has no database at all",[10,1825,1826],{},"Not every source is a table. Rates, holidays, product metadata and half the reference data a report needs live behind an HTTP API, and the pattern is the same: fetch, normalise into a DataFrame, then either write it to Excel or store it alongside the query results.",[168,1828,1830],{"className":170,"code":1829,"language":172,"meta":173,"style":173},"import requests\n\nresp = requests.get(\"https:\u002F\u002Fapi.example.com\u002Fv1\u002Ffx\", timeout=30,\n                    headers={\"Authorization\": f\"Bearer {os.environ['FX_TOKEN']}\"})\nresp.raise_for_status()\nrates = pd.json_normalize(resp.json()[\"rates\"])\n",[175,1831,1832,1839,1843,1868,1903,1908],{"__ignoreMap":173},[178,1833,1834,1836],{"class":180,"line":181},[178,1835,192],{"class":184},[178,1837,1838],{"class":188}," requests\n",[178,1840,1841],{"class":180,"line":198},[178,1842,202],{"emptyLinePlaceholder":201},[178,1844,1845,1848,1850,1853,1856,1858,1861,1863,1866],{"class":180,"line":205},[178,1846,1847],{"class":188},"resp ",[178,1849,236],{"class":184},[178,1851,1852],{"class":188}," requests.get(",[178,1854,1855],{"class":242},"\"https:\u002F\u002Fapi.example.com\u002Fv1\u002Ffx\"",[178,1857,422],{"class":188},[178,1859,1860],{"class":249},"timeout",[178,1862,236],{"class":184},[178,1864,1865],{"class":255},"30",[178,1867,462],{"class":188},[178,1869,1870,1873,1875,1877,1880,1882,1884,1887,1889,1892,1895,1897,1899,1901],{"class":180,"line":212},[178,1871,1872],{"class":249},"                    headers",[178,1874,236],{"class":184},[178,1876,410],{"class":188},[178,1878,1879],{"class":242},"\"Authorization\"",[178,1881,416],{"class":188},[178,1883,885],{"class":184},[178,1885,1886],{"class":242},"\"Bearer ",[178,1888,410],{"class":763},[178,1890,1891],{"class":188},"os.environ[",[178,1893,1894],{"class":242},"'FX_TOKEN'",[178,1896,1607],{"class":188},[178,1898,769],{"class":763},[178,1900,760],{"class":242},[178,1902,433],{"class":188},[178,1904,1905],{"class":180,"line":218},[178,1906,1907],{"class":188},"resp.raise_for_status()\n",[178,1909,1910,1913,1915,1918,1921],{"class":180,"line":224},[178,1911,1912],{"class":188},"rates ",[178,1914,236],{"class":184},[178,1916,1917],{"class":188}," pd.json_normalize(resp.json()[",[178,1919,1920],{"class":242},"\"rates\"",[178,1922,1923],{"class":188},"])\n",[10,1925,1926,1929,1930,1932,1933,1937],{},[175,1927,1928],{},"raise_for_status()"," and an explicit ",[175,1931,1860],{}," are the two lines people leave out and then debug at 03:00: without the timeout a hung API hangs the report forever, and without the status check an HTML error page is parsed as data. ",[17,1934,1936],{"href":1935},"\u002Fadvanced-data-transformation-and-cleaning\u002Fmoving-data-between-excel-and-databases\u002Ffetch-api-data-into-excel-with-python-requests\u002F","Fetch API Data into Excel with Python requests"," covers pagination, retries and flattening nested JSON into columns.",[160,1939,1941],{"id":1940},"refresh-on-a-schedule-not-on-request","Refresh on a schedule, not on request",[10,1943,1944,1945,1949],{},"Once the export works, the last step is making it happen without anyone asking. That is ",[17,1946,1948],{"href":1947},"\u002Fautomating-reporting-workflows\u002Fscheduling-python-excel-scripts-with-cron\u002F","scheduling"," work rather than database work, with one addition specific to this topic: a refreshed extract should be written atomically, so a reader who opens the file mid-refresh never sees a half-written workbook.",[168,1951,1953],{"className":170,"code":1952,"language":172,"meta":173,"style":173},"tmp = target.with_suffix(\".tmp.xlsx\")\nexport(df, tmp)\nos.replace(tmp, target)        # atomic on the same filesystem\n",[175,1954,1955,1970,1975],{"__ignoreMap":173},[178,1956,1957,1960,1962,1965,1968],{"class":180,"line":181},[178,1958,1959],{"class":188},"tmp ",[178,1961,236],{"class":184},[178,1963,1964],{"class":188}," target.with_suffix(",[178,1966,1967],{"class":242},"\".tmp.xlsx\"",[178,1969,259],{"class":188},[178,1971,1972],{"class":180,"line":198},[178,1973,1974],{"class":188},"export(df, tmp)\n",[178,1976,1977,1980],{"class":180,"line":205},[178,1978,1979],{"class":188},"os.replace(tmp, target)        ",[178,1981,1982],{"class":208},"# atomic on the same filesystem\n",[10,1984,1985],{},"Keep the schedule itself boring. A refresh that runs a few minutes after the source system's own load window finishes will be right almost every morning and wrong on the one day the load is late — so where the source publishes a completion marker, wait for it rather than for a clock. Where it does not, check the freshness of the newest row before publishing, and skip the run rather than republish yesterday's numbers under today's timestamp.",[10,1987,1988,1989,1993],{},"Add a small \"generated at\" cell on the summary sheet while you are there. A report that does not say when it was produced gets treated as current forever, which is how a Monday morning decision ends up being made on Thursday's numbers. ",[17,1990,1992],{"href":1991},"\u002Fadvanced-data-transformation-and-cleaning\u002Fmoving-data-between-excel-and-databases\u002Frefresh-an-excel-report-from-a-database-on-a-schedule\u002F","Refresh an Excel Report from a Database on a Schedule"," covers incremental refreshes and the freshness stamp.",[160,1995,1997],{"id":1996},"key-takeaways","Key takeaways",[1999,2000,2001,2012,2018,2027,2033,2048],"ul",{},[2002,2003,2004,2008,2009,2011],"li",{},[2005,2006,2007],"strong",{},"One engine, one URL, from the environment."," SQLAlchemy gives every database the same interface; the password belongs in an environment variable and ",[175,2010,265],{}," keeps an overnight job's connection usable.",[2002,2013,2014,2017],{},[2005,2015,2016],{},"Bind parameters, never f-strings."," Values that came from a spreadsheet must not be able to change the meaning of a statement.",[2002,2019,2020,2023,2024,2026],{},[2005,2021,2022],{},"Stage the inbound direction."," Load to a staging table, validate there, then merge inside ",[175,2025,987],{}," so a bad file cannot reach readers.",[2002,2028,2029,2032],{},[2005,2030,2031],{},"Make loads idempotent."," Delete the period and re-insert, so a retry produces the same table rather than duplicated rows.",[2002,2034,2035,2038,2039,2041,2042,2044,2045,2047],{},[2005,2036,2037],{},"Be explicit about types at the read."," Identifiers as ",[175,2040,1418],{},", nullable integers as ",[175,2043,1414],{},", dates coerced with a ",[175,2046,1440],{}," check, money rounded before it is written.",[2002,2049,2050,2053,2054,2057],{},[2005,2051,2052],{},"Write extracts atomically and stamp them."," A temporary file plus ",[175,2055,2056],{},"os.replace",", and a visible \"generated at\" so nobody mistakes a stale workbook for a current one.",[160,2059,2061],{"id":2060},"frequently-asked-questions","Frequently asked questions",[10,2063,2064,2067,2068,2070],{},[2005,2065,2066],{},"Do I need SQLAlchemy, or can pandas talk to the database directly?","\npandas accepts a raw DBAPI connection for reading, but ",[175,2069,145],{}," officially supports SQLAlchemy connectables and SQLite connections only. Since SQLAlchemy also gives you one URL format for every database and safe parameter binding, use it for anything beyond a throwaway script.",[10,2072,2073,2076],{},[2005,2074,2075],{},"How do I stop a spreadsheet load from corrupting a live table?","\nLoad into a staging table first, validate it there, then move the rows across in one transaction. Writing straight into the reporting table means a bad file is visible to readers before anyone notices.",[10,2078,2079,2082,2083,2085,2086,2089,2090,2093,2094,2096],{},[2005,2080,2081],{},"Why do my IDs come back as 1.0 instead of 1?","\nA column with any blank cell becomes ",[175,2084,1400],{}," in pandas, because ",[175,2087,2088],{},"NaN"," cannot live in an int column. Read the column with ",[175,2091,2092],{},"dtype=str",", or use pandas' nullable ",[175,2095,1414],{}," type.",[10,2098,2099,2102,2103,2106,2107,2110],{},[2005,2100,2101],{},"Is it safe to build the SQL with an f-string?","\nNo. Use bound parameters — SQLAlchemy's ",[175,2104,2105],{},"text()"," with ",[175,2108,2109],{},":name"," placeholders — so a value taken from a spreadsheet cell cannot change the meaning of the statement.",[10,2112,2113,2116],{},[2005,2114,2115],{},"How large can an export be before Excel is the wrong answer?","\nExcel's hard limit is 1,048,576 rows per sheet, but the practical limit is far lower: a workbook past a few hundred thousand rows is slow to open and slower to filter. Beyond that, export CSV or Parquet and keep Excel for the summary.",[160,2118,2120],{"id":2119},"related","Related",[1999,2122,2123,2132,2146,2161,2171],{},[2002,2124,2125,2128,2129,2131],{},[2005,2126,2127],{},"Parent:"," ",[17,2130,20],{"href":19}," — the cleaning and validation this boundary depends on.",[2002,2133,2134,2128,2137,422,2139,422,2141,565,2143,2145],{},[2005,2135,2136],{},"In this topic:",[17,2138,580],{"href":579},[17,2140,997],{"href":996},[17,2142,1936],{"href":1935},[17,2144,1992],{"href":1991},".",[2002,2147,2148,2128,2151,2155,2156,2160],{},[2005,2149,2150],{},"Sibling topics:",[17,2152,2154],{"href":2153},"\u002Fadvanced-data-transformation-and-cleaning\u002Fvalidating-excel-data-with-python\u002F","Validating Excel Data with Python"," for the checks a staging table should run, and ",[17,2157,2159],{"href":2158},"\u002Fadvanced-data-transformation-and-cleaning\u002Fworking-with-large-excel-files-in-python\u002F","Working with Large Excel Files in Python"," when the extract outgrows a comfortable workbook.",[2002,2162,2163,2128,2166,2170],{},[2005,2164,2165],{},"Downstream:",[17,2167,2169],{"href":2168},"\u002Fautomating-reporting-workflows\u002F","Automating Reporting Workflows"," — scheduling, delivering and monitoring the refresh.",[2002,2172,2173,2128,2176,2178],{},[2005,2174,2175],{},"Presentation:",[17,2177,284],{"href":283}," — turning a query result into a workbook people will actually read.",[2180,2181,2182],"style",{},"html pre.shiki code .s-kum, html code.shiki .s-kum{--shiki-default:#D73A49;--shiki-dark:#FF9492}html pre.shiki code .skGVy, html code.shiki .skGVy{--shiki-default:#24292E;--shiki-dark:#F0F3F6}html pre.shiki code .s-wDw, html code.shiki .s-wDw{--shiki-default:#6A737D;--shiki-dark:#BDC4CC}html pre.shiki code .srMev, html code.shiki .srMev{--shiki-default:#032F62;--shiki-dark:#ADDCFF}html pre.shiki code .sa561, html code.shiki .sa561{--shiki-default:#E36209;--shiki-dark:#FFB757}html pre.shiki code .sP0c6, html code.shiki .sP0c6{--shiki-default:#005CC5;--shiki-dark:#91CBFF}html .default .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .dark .shiki span {color: var(--shiki-dark);background: var(--shiki-dark-bg);font-style: var(--shiki-dark-font-style);font-weight: var(--shiki-dark-font-weight);text-decoration: var(--shiki-dark-text-decoration);}html.dark .shiki span {color: var(--shiki-dark);background: var(--shiki-dark-bg);font-style: var(--shiki-dark-font-style);font-weight: var(--shiki-dark-font-weight);text-decoration: var(--shiki-dark-text-decoration);}html pre.shiki code .s_Opv, html code.shiki .s_Opv{--shiki-default:#6F42C1;--shiki-dark:#DBB7FF}html pre.shiki code .sSjpA, html code.shiki .sSjpA{--shiki-default:#005CC5;--shiki-dark:#FF9492}",{"title":173,"searchDepth":198,"depth":198,"links":2184},[2185,2186,2187,2188,2189,2190,2191,2192,2193,2194,2195,2196],{"id":162,"depth":198,"text":163},{"id":272,"depth":198,"text":273},{"id":584,"depth":198,"text":585},{"id":1001,"depth":198,"text":1002},{"id":1153,"depth":198,"text":1154},{"id":1345,"depth":198,"text":1346},{"id":1471,"depth":198,"text":1472},{"id":1822,"depth":198,"text":1823},{"id":1940,"depth":198,"text":1941},{"id":1996,"depth":198,"text":1997},{"id":2060,"depth":198,"text":2061},{"id":2119,"depth":198,"text":2120},"2026-08-10","Excel as the interface, the database as the source of truth: query to workbook with SQLAlchemy, load a spreadsheet into a table safely, keep types intact in both directions, and refresh on a schedule.","md",[2201,2203,2205,2207,2209],{"q":2066,"a":2202},"pandas accepts a raw DBAPI connection for reading, but to_sql officially supports SQLAlchemy connectables and SQLite connections only. Since SQLAlchemy also gives you one URL format for every database and safe parameter binding, use it for anything beyond a throwaway script.",{"q":2075,"a":2204},"Load into a staging table first, validate it there, then move the rows across in one transaction. Writing straight into the reporting table means a bad file is visible to readers before anyone notices.",{"q":2081,"a":2206},"A column with any blank cell becomes float64 in pandas, because NaN cannot live in an int column. Read the column with dtype=str, or use pandas' nullable Int64 type.",{"q":2101,"a":2208},"No. Use bound parameters — SQLAlchemy's text() with :name placeholders — so a value taken from a spreadsheet cell cannot change the meaning of the statement.",{"q":2115,"a":2210},{"Excel's hard limit is 1,048,576 rows per sheet, but the practical limit is far lower":2211},"a workbook past a few hundred thousand rows is slow to open and slower to filter. Beyond that, export CSV or Parquet and keep Excel for the summary.",{"breadcrumb":2213},[2214,2216,2217],{"name":2215,"item":1766},"Home",{"name":20,"item":19},{"name":5,"item":2218},"\u002Fadvanced-data-transformation-and-cleaning\u002Fmoving-data-between-excel-and-databases\u002F","\u002Fadvanced-data-transformation-and-cleaning\u002Fmoving-data-between-excel-and-databases",{"title":2221,"description":2222},"Move Data Between Excel and SQL with Python","Export SQL query results to Excel and load spreadsheets into a database with pandas and SQLAlchemy — connections, chunking, type mapping, staging tables and scheduled refreshes.","moving-data-between-excel-and-databases","advanced-data-transformation-and-cleaning\u002Fmoving-data-between-excel-and-databases\u002Findex","guide","jpYLhh_2CFKUUje-iHUz2wJp84dF4iomyRRNEd1mODs",[2228,2232],{"title":2229,"path":2230,"stem":2231,"children":-1},"The VLOOKUP Equivalent in pandas for Excel Files","\u002Fadvanced-data-transformation-and-cleaning\u002Fmerging-and-joining-excel-dataframes\u002Fvlookup-equivalent-in-pandas-for-excel-files","advanced-data-transformation-and-cleaning\u002Fmerging-and-joining-excel-dataframes\u002Fvlookup-equivalent-in-pandas-for-excel-files\u002Findex",{"title":580,"path":2233,"stem":2234,"children":-1},"\u002Fadvanced-data-transformation-and-cleaning\u002Fmoving-data-between-excel-and-databases\u002Fexport-sql-query-results-to-excel-with-python","advanced-data-transformation-and-cleaning\u002Fmoving-data-between-excel-and-databases\u002Fexport-sql-query-results-to-excel-with-python\u002Findex",1786800026511]