diff --git a/Excel-CSV-File-Merger-main/LICENSE b/Excel-CSV-File-Merger-main/LICENSE new file mode 100644 index 0000000..b04d288 --- /dev/null +++ b/Excel-CSV-File-Merger-main/LICENSE @@ -0,0 +1,21 @@ +MIT License + +Copyright (c) 2026 Tuff-Tech-coder + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. diff --git a/Excel-CSV-File-Merger-main/README.md b/Excel-CSV-File-Merger-main/README.md new file mode 100644 index 0000000..c00b8fa --- /dev/null +++ b/Excel-CSV-File-Merger-main/README.md @@ -0,0 +1,157 @@ +# Excel & CSV File Merger + +![Python](https://img.shields.io/badge/Python-3.11+-3776AB?logo=python&logoColor=white) +![pandas](https://img.shields.io/badge/pandas-3.x-150458?logo=pandas&logoColor=white) +![openpyxl](https://img.shields.io/badge/openpyxl-3.x-1D6F42) +![Tests](https://img.shields.io/badge/tests-34%20passing-brightgreen) +![License](https://img.shields.io/badge/License-MIT-green) + +A data-normalization pipeline that scans a folder for Excel and CSV files and intelligently merges them into a single, clean master workbook. Built specifically for the **messy reality** of business data: mismatched column orders, inconsistent naming, currency stored as text, stray whitespace, blank rows, and duplicates. + +--- + +## Why it's useful + +Combining spreadsheets from different teams by hand is slow and error-prone — column names never match, totals are formatted as `"$5,400.00"`, and duplicates creep in. This tool encodes those fixes once and applies them consistently, turning a pile of inconsistent files into one analysis-ready dataset with an audit trail. + +--- + +## Features + +- **Auto-discovery** of every `.xlsx`, `.xls`, and `.csv` file in a folder. +- **Column-alias mapping** — a configurable dictionary unifies variants (`"Sales Amount"`, `"Amount"`, `"Salary"` → `revenue`; `"Territory"` → `region`) so files merge on meaning, not exact spelling. +- **Currency normalization** — `"$5,400.00"` becomes `5400.0` for real math. +- **Cleanup pass** — strips cell whitespace and drops entirely empty rows. +- **Graceful column alignment** — files missing columns still merge cleanly, with gaps filled rather than erroring. +- **Configurable deduplication** of identical rows (`--no-dedup` to disable). +- **Source tracking** — a `source_file` column records each row's origin. +- **Polished two-sheet output** — a formatted *Merged Data* sheet (styled header, alternating row shading, auto-fit columns) plus a *Merge Summary* sheet with per-file counts, duplicates removed, and final totals. + +--- + +## Tech stack + +`Python` · `pandas` · `openpyxl` · `xlrd` · `argparse` · `logging` · `regex` + +--- + +## Project structure + +``` +├── excel_merger.py # Main script +├── create_samples.py # Generates the demo input files +├── requirements.txt # Python dependencies +├── tests/ # 26 unit tests +├── merged_master.xlsx # Generated: merged output (gitignored) +├── merger.log # Generated: application log +└── sample_input/ + ├── sales_q1.xlsx # Standard sales data + ├── sales_q2.xlsx # Different column order + extra column + a duplicate + ├── hr_employees.csv # Missing columns, different naming + ├── marketing_leads.csv # Currency-as-text, whitespace, varied date formats + └── ops_data.xlsx # Renamed columns + blank rows +``` + +--- + +## Setup + +```bash +python -m venv venv +source venv/bin/activate # Windows: venv\Scripts\activate +pip install -r requirements.txt +``` + +--- + +## Usage + +```bash +# Merge the included sample files +python excel_merger.py --input ./sample_input --output merged_master.xlsx + +# Merge your own folder, keep duplicates +python excel_merger.py --input /path/to/folder --no-dedup +``` + +**Options:** `--input` · `--output` · `--no-dedup` + +--- + +## How it works + +1. Discovers all supported files in the input folder. +2. Loads each, normalizes column names through the alias map, cleans cells, normalizes currency, drops empty rows, and tags rows with their source file. +3. Concatenates everything (pandas aligns on column name, filling gaps). +4. Optionally deduplicates on all columns except the source tag. +5. Writes a formatted workbook with both the merged data and a summary sheet. + +--- + +## Customizing: the `COLUMN_ALIASES` extension point + +This is the most reusable idea in the repo. `COLUMN_ALIASES` is a single +dictionary mapping every known source header onto a canonical name: + +```python +COLUMN_ALIASES: dict[str, str] = { + "full name": "name", + "employee name": "name", + "sales amount": "revenue", + "amount": "revenue", + "salary": "revenue", + "territory": "region", + ... +} +``` + +Teaching the merger a new file format is **one line in this dict** — no other +code changes. Everything downstream (currency normalization, deduplication, +column alignment, the summary sheet) keys off canonical names, so the entire +pipeline picks up the new variant automatically. Schema drift becomes +configuration rather than a code change. + +### Two caveats worth knowing + +**Aliasing can collide.** Several headers intentionally map to the same +canonical name — `Amount`, `Total` and `Salary` all become `revenue`. If a +*single file* contains two of them, they cannot both be `revenue`. `dedupe_columns()` +renames the second to `revenue_2` and logs a warning naming the file: + +``` +marketing_leads.csv: duplicate canonical column 'revenue' renamed to 'revenue_2'. +Review COLUMN_ALIASES if these should be merged. +``` + +Nothing is silently dropped, and nothing crashes — you get both columns plus a +prompt to decide whether that mapping was right for your data. Note that only +the canonical `revenue` column gets currency normalization; `revenue_2` is left +as-is precisely because the tool should not guess which one you meant. + +**`"first name" → "name"` is lossy.** If a file has separate `First Name` and +`Last Name` columns, only the first is mapped to `name` and the surname stays +under its own column. That is a deliberate simplification, not a bug — but if +your data splits names that way, either remove that alias or pre-join the two +columns before merging. + +--- + +## Development + +```bash +pip install -r requirements.txt +pip install pytest ruff + +pytest -q # 34 tests +ruff check . +``` + +The suite covers alias normalization, currency parsing, whitespace cleaning, +deduplication and source-file tagging, including a regression test for the +alias-collision crash described above. + +--- + +## Possible extensions + +Add per-column type coercion, configurable merge keys for true joins (not just concatenation), a dry-run preview, or output to Parquet/SQL for larger datasets. diff --git a/Excel-CSV-File-Merger-main/__pycache__/create_samples.cpython-312.pyc b/Excel-CSV-File-Merger-main/__pycache__/create_samples.cpython-312.pyc new file mode 100644 index 0000000..d3ebf68 Binary files /dev/null and b/Excel-CSV-File-Merger-main/__pycache__/create_samples.cpython-312.pyc differ diff --git a/Excel-CSV-File-Merger-main/__pycache__/excel_merger.cpython-312.pyc b/Excel-CSV-File-Merger-main/__pycache__/excel_merger.cpython-312.pyc new file mode 100644 index 0000000..3c29792 Binary files /dev/null and b/Excel-CSV-File-Merger-main/__pycache__/excel_merger.cpython-312.pyc differ diff --git a/Excel-CSV-File-Merger-main/create_samples.py b/Excel-CSV-File-Merger-main/create_samples.py new file mode 100644 index 0000000..dd60762 --- /dev/null +++ b/Excel-CSV-File-Merger-main/create_samples.py @@ -0,0 +1,133 @@ +"""Create five deliberately messy sample input files for the Excel Merger demo.""" + +import csv +from pathlib import Path + +import pandas as pd + +OUT = Path(__file__).parent / "sample_input" + + +def create_samples(output_folder: Path = OUT) -> None: + """Generate the demo workbooks and CSV files in ``output_folder``.""" + output_folder.mkdir(parents=True, exist_ok=True) + + # ── File 1: sales_q1.xlsx ───────────────────────────────────────────── + # Standard sales data, some mixed-case headers + df1 = pd.DataFrame({ + "Name": ["Alice Johnson", "Bob Smith", "Carol White", "Dave Brown", "Eve Davis"], + "Email": [ + "alice@corp.com", + "bob@corp.com", + "carol@corp.com", + "dave@corp.com", + "eve@corp.com", + ], + "Revenue": [12500.00, 8750.50, 23400.00, 5600.75, 19800.00], + "Region": ["North", "South", "North", "West", "East"], + "Date": ["2024-01-15", "2024-01-20", "2024-02-03", "2024-02-14", "2024-03-01"], + "Department": ["Sales", "Sales", "Sales", "Sales", "Sales"], + }) + df1.to_excel(output_folder / "sales_q1.xlsx", index=False) + + # ── File 2: sales_q2.xlsx ───────────────────────────────────────────── + # Same data type but different column order and some extra columns + df2 = pd.DataFrame({ + "email": [ + "frank@corp.com", + "grace@corp.com", + "henry@corp.com", + "irene@corp.com", + "jake@corp.com", + "alice@corp.com", + ], + "REVENUE": [7200.00, 15300.00, 9800.00, 22100.00, 6400.00, 12500.00], + "Region": ["South", "East", "North", "West", "South", "North"], + "name": ["Frank Lee", "Grace Kim", "Henry Chen", "Irene Park", "Jake Wu", "Alice Johnson"], + "Date": [ + "2024-04-10", + "2024-04-22", + "2024-05-08", + "2024-05-19", + "2024-06-01", + "2024-01-15", + ], + "Notes": ["Top performer", "", "Needs review", "Excellent", "", "Duplicate entry"], + "Department": ["Sales", "Sales", "Sales", "Sales", "Sales", "Sales"], + }) + df2.to_excel(output_folder / "sales_q2.xlsx", index=False) + + # ── File 3: hr_employees.csv ───────────────────────────────────────── + # CSV format, missing Revenue and Date columns, extra HR-specific columns + df3 = pd.DataFrame({ + "Name": [ + "Laura Moss", + "Mike Stone", + "Nancy Hill", + "Oscar Reed", + "Paula Marsh", + "Quinn Ford", + "Rachel Moore", + ], + "Email": [ + "laura@corp.com", + "mike@corp.com", + "nancy@corp.com", + "oscar@corp.com", + "paula@corp.com", + "quinn@corp.com", + "rachel@corp.com", + ], + "Department": ["HR", "HR", "Engineering", "Engineering", "Marketing", "Marketing", "HR"], + "Region": ["East", "West", "North", "South", "East", "West", "North"], + "Hire Date": [ + "2021-03-15", + "2019-07-01", + "2022-11-20", + "2020-04-05", + "2023-01-10", + "2018-09-30", + "2024-02-15", + ], + "Salary": [55000, 72000, 98000, 115000, 61000, 68000, 52000], + }) + df3.to_csv(output_folder / "hr_employees.csv", index=False) + + # ── File 4: marketing_leads.csv ────────────────────────────────────── + # CSV with inconsistent formatting: extra whitespace, mixed number formats + rows = [ + [" Name ", "Email", "Revenue", " Region ", "Date", "Department"], + ["Sam Turner ", "sam@leads.com", "$5,400.00", " North ", "2024/01/08", "Marketing"], + [" Tina Brooks", "tina@leads.com", "8200", "East", "2024-02-14", "Marketing"], + ["Uma Patel ", "uma@leads.com", "$11,750.50", "West ", "March 5, 2024", "Marketing"], + ["Vince Hall", "vince@leads.com", "3100.0", "South", "2024-04-22", "Marketing"], + [" Wendy Fox", "wendy@leads.com", "$19,900", "North", "2024-05-30", "Marketing"], + ["Xavier Long", "xavier@leads.com", "7650.25", " East", "2024-06-15", "Marketing"], + ] + with (output_folder / "marketing_leads.csv").open("w", encoding="utf-8", newline="") as f: + writer = csv.writer(f) + writer.writerows(rows) + + # ── File 5: ops_data.xlsx ──────────────────────────────────────────── + # Different column names, some blank rows, purely numeric revenue + df5 = pd.DataFrame({ + "Full Name": ["Yara Singh", "Zach Adams", None, "Amy Clarke", "Brian Duke"], + "Contact Email": [ + "yara@ops.com", + "zach@ops.com", + None, + "amy@ops.com", + "brian@ops.com", + ], + "Sales Amount": [14200.00, 9900.50, None, 31000.00, 7800.25], + "Territory": ["West", "North", None, "South", "East"], + "Transaction Date": ["2024-03-18", "2024-04-02", None, "2024-05-11", "2024-06-29"], + "Team": ["Operations", "Operations", None, "Operations", "Operations"], + }) + df5.to_excel(output_folder / "ops_data.xlsx", index=False) + + print(f"[OK] Created 5 sample input files in {output_folder}") + + +if __name__ == "__main__": + create_samples() diff --git a/Excel-CSV-File-Merger-main/excel_merger.py b/Excel-CSV-File-Merger-main/excel_merger.py new file mode 100644 index 0000000..88fe8ae --- /dev/null +++ b/Excel-CSV-File-Merger-main/excel_merger.py @@ -0,0 +1,418 @@ +""" +Excel & CSV File Merger +======================= +Scans a folder for all Excel (.xlsx, .xls) and CSV files, intelligently +merges them into a single normalized master file, and outputs a clean Excel +file. Handles messy real-world data: different column orders, missing columns, +duplicate rows, inconsistent formatting, and varied column naming conventions. + +Usage: + python excel_merger.py --input ./sample_input --output merged_master.xlsx + python excel_merger.py --input /path/to/folder --no-dedup + +Features: + - Discovers all .xlsx, .xls, and .csv files recursively + - Normalizes column names (strips whitespace, lowercases for matching) + - Maps common column aliases (e.g. "Sales Amount" → "revenue") + - Fills missing columns with empty values + - Deduplicates identical rows (configurable) + - Strips leading/trailing whitespace from all text cells + - Normalizes currency strings ("$5,400.00" → 5400.00) + - Adds a "source_file" column tracking which file each row came from + - Outputs a formatted Excel file with statistics summary sheet +""" + +import argparse +import logging +import re +from pathlib import Path + +import pandas as pd +from openpyxl.styles import Alignment, Font, PatternFill + +logger = logging.getLogger(__name__) + +SUPPORTED_SUFFIXES = {".xlsx", ".xls", ".csv"} +SOURCE_COLUMN = "source_file" + + +# --------------------------------------------------------------------------- +# Logging +# --------------------------------------------------------------------------- +def configure_logging(log_path: Path = Path("merger.log")) -> None: + """Configure console and file logging for the command-line entry point. + + Logging is deliberately configured at runtime rather than import time so + importing this module as a library neither creates a file nor fails merely + because the current directory is read-only. + """ + handlers: list[logging.Handler] = [logging.StreamHandler()] + file_error: OSError | None = None + try: + # Input files are arbitrary user data, so the log is explicitly UTF-8 + # rather than the platform default (cp1252 on Windows). + handlers.append(logging.FileHandler(log_path, encoding="utf-8")) + except OSError as exc: + file_error = exc + + logging.basicConfig( + level=logging.INFO, + format="%(asctime)s [%(levelname)s] %(message)s", + handlers=handlers, + force=True, + ) + if file_error is not None: + logger.warning("Could not write log file %s: %s", log_path, file_error) + +# --------------------------------------------------------------------------- +# Column alias map — maps known variants to a canonical name +# Add more entries here as you encounter new file formats. +# --------------------------------------------------------------------------- +COLUMN_ALIASES: dict[str, str] = { + # Name variants + "full name": "name", + "full_name": "name", + "first name": "name", + "employee name": "name", + # Email variants + "contact email": "email", + "email address": "email", + "e-mail": "email", + # Revenue / money variants + "revenue": "revenue", + "sales amount": "revenue", + "sales_amount": "revenue", + "amount": "revenue", + "salary": "revenue", + "total": "revenue", + # Region / territory variants + "region": "region", + "territory": "region", + "area": "region", + "zone": "region", + # Date variants + "date": "date", + "transaction date": "date", + "transaction_date": "date", + "hire date": "date", + "hire_date": "date", + "sale date": "date", + # Department variants + "department": "department", + "dept": "department", + "team": "department", + "division": "department", +} + + +def normalize_column_name(col: str) -> str: + """ + Lowercase, strip whitespace, and apply alias mapping. + Returns the canonical column name. + """ + cleaned = str(col).strip().lower() + return COLUMN_ALIASES.get(cleaned, cleaned) + + +def dedupe_columns(cols: list[str], source_filename: str) -> list[str]: + """ + Ensure canonical column names are unique within a single file. + + Multiple source headers can alias to the same canonical name (e.g. both + "Amount" and "Total" map to "revenue"). Left unhandled, pandas produces two + identically-named columns, `df["revenue"]` returns a DataFrame instead of a + Series, and every downstream scalar operation raises + "The truth value of a Series is ambiguous". + """ + used: set[str] = set() + next_suffix: dict[str, int] = {} + out: list[str] = [] + for c in cols: + renamed = c + if renamed in used: + suffix = next_suffix.get(c, 2) + renamed = f"{c}_{suffix}" + while renamed in used: + suffix += 1 + renamed = f"{c}_{suffix}" + next_suffix[c] = suffix + 1 + logger.warning( + f"{source_filename}: duplicate canonical column '{c}' renamed to " + f"'{renamed}'. Review COLUMN_ALIASES if these should be merged." + ) + used.add(renamed) + next_suffix.setdefault(c, 2) + out.append(renamed) + return out + + +def normalize_currency(value) -> float | None: + """ + Convert currency strings like '$5,400.00' or '5400' to float. + Returns None if conversion is not possible. + """ + if isinstance(value, (pd.Series, pd.DataFrame)): + raise TypeError( + "normalize_currency expects a scalar. Receiving a Series means the " + "DataFrame has duplicate column names -- run dedupe_columns() first." + ) + if pd.isna(value): + return None + raw = str(value).strip() + # Remove currency symbols and thousands separators + cleaned = re.sub(r"[^\d.\-]", "", raw) + try: + return float(cleaned) if cleaned else None + except ValueError: + return None + + +def clean_string(value) -> str: + """Strip whitespace from string values; leave non-strings unchanged.""" + if isinstance(value, str): + return value.strip() + return value + + +# --------------------------------------------------------------------------- +# File loading +# --------------------------------------------------------------------------- +def load_file(path: Path) -> pd.DataFrame | None: + """ + Load an Excel or CSV file into a DataFrame. + Returns None if the file cannot be read. + """ + try: + suffix = path.suffix.lower() + if suffix in (".xlsx", ".xls"): + df = pd.read_excel(path, engine="openpyxl" if suffix == ".xlsx" else "xlrd") + elif suffix == ".csv": + df = pd.read_csv(path) + else: + logger.warning(f"Unsupported file type: {path.name}") + return None + + logger.info(f"Loaded {path.name}: {len(df)} rows, {len(df.columns)} columns") + return df + + except Exception as e: + logger.error(f"Failed to load {path.name}: {e}") + return None + + +# --------------------------------------------------------------------------- +# Per-file normalization +# --------------------------------------------------------------------------- +def normalize_dataframe(df: pd.DataFrame, source_filename: str) -> pd.DataFrame: + """ + Apply all normalization steps to a single DataFrame: + 1. Normalize column names (strip + alias mapping) + 2. Strip whitespace from string cells + 3. Drop entirely empty rows + 4. Normalize currency values in 'revenue' column + 5. Add source_file column + """ + # Work on a copy so callers do not get partially mutated input if a later + # normalization step fails. + df = df.copy() + + # --- 1. Normalize column names --- + # Reserve SOURCE_COLUMN before normalizing user columns. Otherwise an input + # column named "source_file" would be silently overwritten in step 5. + normalized_columns = [normalize_column_name(c) for c in df.columns] + df.columns = dedupe_columns( + [SOURCE_COLUMN, *normalized_columns], source_filename + )[1:] + + # --- 2. Strip whitespace from all string cells --- + df = df.map(clean_string) + + # --- 3. Drop rows where ALL values are missing or empty after trimming --- + # dropna alone does not consider "" empty, so a row made entirely of + # whitespace survived step 2 in earlier versions. + empty_cells = df.isna() | df.eq("") + df = df.loc[~empty_cells.all(axis=1)].copy() + + # --- 4. Normalize revenue column --- + if "revenue" in df.columns: + df["revenue"] = df["revenue"].apply(normalize_currency) + + # --- 5. Tag with source --- + df[SOURCE_COLUMN] = source_filename + + return df + + +# --------------------------------------------------------------------------- +# Merge engine +# --------------------------------------------------------------------------- +def merge_files( + input_folder: Path, + output_path: Path, + deduplicate: bool = True, +) -> None: + """ + Main merge function. Discovers files, normalizes each, aligns columns, + concatenates, deduplicates, and writes to Excel. + """ + # --- Discover all supported files --- + # Inspect suffixes case-insensitively so names such as DATA.CSV work on + # case-sensitive platforms too. The requested output is excluded so a + # rerun cannot ingest its own previous result. + output_resolved = output_path.resolve() + found_files = sorted( + ( + path + for path in input_folder.rglob("*") + if path.is_file() + and path.suffix.lower() in SUPPORTED_SUFFIXES + and not path.name.startswith("~$") + and path.resolve() != output_resolved + ), + key=lambda path: path.relative_to(input_folder).as_posix().casefold(), + ) + + if not found_files: + raise FileNotFoundError(f"No Excel or CSV files found in: {input_folder}") + + discovered_names = [path.relative_to(input_folder).as_posix() for path in found_files] + logger.info(f"Found {len(found_files)} file(s) to merge: {discovered_names}") + + # --- Load and normalize each file --- + frames: list[pd.DataFrame] = [] + load_stats: list[dict] = [] + + for path in found_files: + source_filename = path.relative_to(input_folder).as_posix() + df = load_file(path) + if df is None: + continue + original_rows = len(df) + df = normalize_dataframe(df, source_filename) + frames.append(df) + load_stats.append({ + "file": source_filename, + "rows_loaded": original_rows, + "columns": len(df.columns) - 1, # exclude source_file + }) + + if not frames: + raise ValueError("No files could be successfully loaded") + + # --- Concatenate (pandas aligns on column names, fills missing with NaN) --- + merged = pd.concat(frames, ignore_index=True, sort=False) + rows_before_dedup = len(merged) + logger.info(f"Combined total: {rows_before_dedup} rows") + + # --- Deduplicate (exclude source_file from key) --- + dupes_removed = 0 + if deduplicate: + key_cols = [c for c in merged.columns if c != SOURCE_COLUMN] + merged.drop_duplicates(subset=key_cols, keep="first", inplace=True) + merged.reset_index(drop=True, inplace=True) + dupes_removed = rows_before_dedup - len(merged) + if dupes_removed: + logger.info(f"Removed {dupes_removed} duplicate row(s)") + + # --- Reorder: put source_file last --- + cols = [c for c in merged.columns if c != SOURCE_COLUMN] + [SOURCE_COLUMN] + merged = merged[cols] + + # --- Write output --- + write_excel(merged, output_path, load_stats, dupes_removed) + logger.info(f"Merge complete: {len(merged)} rows written to {output_path}") + print(f"\n[OK] Merged {len(frames)} files -> {len(merged)} rows -> {output_path}") + + +# --------------------------------------------------------------------------- +# Excel writer with formatting +# --------------------------------------------------------------------------- +def write_excel(df: pd.DataFrame, output_path: Path, stats: list[dict], dupes_removed: int) -> None: + """Write the merged DataFrame plus a summary sheet to a formatted Excel file.""" + output_path.parent.mkdir(parents=True, exist_ok=True) + with pd.ExcelWriter(output_path, engine="openpyxl") as writer: + # --- Main data sheet --- + df.to_excel(writer, index=False, sheet_name="Merged Data") + _format_sheet(writer.sheets["Merged Data"], df) + + # --- Summary sheet --- + summary_rows = [] + for s in stats: + summary_rows.append(s) + summary_rows.append({"file": "─" * 20, "rows_loaded": "─" * 10, "columns": "─" * 10}) + summary_rows.append({ + "file": "TOTAL", + "rows_loaded": sum(s["rows_loaded"] for s in stats), + "columns": "N/A", + }) + summary_rows.append({"file": "Duplicates removed", "rows_loaded": dupes_removed, "columns": ""}) + summary_rows.append({"file": "Final row count", "rows_loaded": len(df), "columns": ""}) + + summary_df = pd.DataFrame(summary_rows, columns=["file", "rows_loaded", "columns"]) + summary_df.columns = ["Source File", "Rows Loaded", "Original Columns"] + summary_df.to_excel(writer, index=False, sheet_name="Merge Summary") + _format_sheet(writer.sheets["Merge Summary"], summary_df) + + +def _format_sheet(ws, df: pd.DataFrame) -> None: + """Apply header styling and auto-fit columns to a worksheet.""" + header_font = Font(bold=True, color="FFFFFF") + header_fill = PatternFill(start_color="1A3A5C", end_color="1A3A5C", fill_type="solid") + + for cell in ws[1]: + cell.font = header_font + cell.fill = header_fill + cell.alignment = Alignment(horizontal="center", wrap_text=True) + + # Auto-fit column widths based on content + for col in ws.columns: + max_len = max( + (len(str(cell.value or "")) for cell in col), + default=10, + ) + ws.column_dimensions[col[0].column_letter].width = min(max_len + 2, 50) + + # Alternate row shading for readability + light_fill = PatternFill(start_color="F0F4F8", end_color="F0F4F8", fill_type="solid") + for row_idx, row in enumerate(ws.iter_rows(min_row=2), start=2): + if row_idx % 2 == 0: + for cell in row: + cell.fill = light_fill + + +# --------------------------------------------------------------------------- +# CLI +# --------------------------------------------------------------------------- +def main(): + configure_logging() + parser = argparse.ArgumentParser(description="Excel & CSV File Merger") + parser.add_argument( + "--input", + default="./sample_input", + help="Folder containing Excel/CSV files to merge (default: ./sample_input)", + ) + parser.add_argument( + "--output", + default="merged_master.xlsx", + help="Output Excel filename (default: merged_master.xlsx)", + ) + parser.add_argument( + "--no-dedup", + action="store_true", + help="Skip duplicate row removal", + ) + args = parser.parse_args() + + input_folder = Path(args.input) + if not input_folder.is_dir(): + raise ValueError(f"Input path is not a directory: {input_folder}") + + merge_files( + input_folder=input_folder, + output_path=Path(args.output), + deduplicate=not args.no_dedup, + ) + + +if __name__ == "__main__": + main() diff --git a/Excel-CSV-File-Merger-main/merger.log b/Excel-CSV-File-Merger-main/merger.log new file mode 100644 index 0000000..34d10c2 --- /dev/null +++ b/Excel-CSV-File-Merger-main/merger.log @@ -0,0 +1,8 @@ +2026-09-26 09:38:40,994 [INFO] Found 5 file(s) to merge: ['hr_employees.csv', 'marketing_leads.csv', 'ops_data.xlsx', 'sales_q1.xlsx', 'sales_q2.xlsx'] +2026-09-26 09:38:40,995 [INFO] Loaded hr_employees.csv: 7 rows, 6 columns +2026-09-26 09:38:40,999 [INFO] Loaded marketing_leads.csv: 6 rows, 6 columns +2026-09-26 09:38:41,004 [INFO] Loaded ops_data.xlsx: 5 rows, 6 columns +2026-09-26 09:38:41,008 [INFO] Loaded sales_q1.xlsx: 5 rows, 6 columns +2026-09-26 09:38:41,013 [INFO] Loaded sales_q2.xlsx: 6 rows, 7 columns +2026-09-26 09:38:41,015 [INFO] Combined total: 28 rows +2026-09-26 09:38:41,039 [INFO] Merge complete: 28 rows written to C:\Users\Dan\Documents\Codex\2026-09-26\cou\work\final-smoke\merged.xlsx diff --git a/Excel-CSV-File-Merger-main/requirements.txt b/Excel-CSV-File-Merger-main/requirements.txt new file mode 100644 index 0000000..2e782a2 --- /dev/null +++ b/Excel-CSV-File-Merger-main/requirements.txt @@ -0,0 +1,3 @@ +pandas==3.0.3 +openpyxl==3.1.5 +xlrd==2.0.2 diff --git a/Excel-CSV-File-Merger-main/ruff.toml b/Excel-CSV-File-Merger-main/ruff.toml new file mode 100644 index 0000000..2095aae --- /dev/null +++ b/Excel-CSV-File-Merger-main/ruff.toml @@ -0,0 +1,9 @@ +line-length = 100 +target-version = "py311" + +[lint] +select = ["E", "F", "I", "UP", "B", "SIM"] +ignore = ["E501"] + +[lint.per-file-ignores] +"tests/*" = ["E402"] diff --git a/Excel-CSV-File-Merger-main/sample_input/hr_employees.csv b/Excel-CSV-File-Merger-main/sample_input/hr_employees.csv new file mode 100644 index 0000000..2ea6620 --- /dev/null +++ b/Excel-CSV-File-Merger-main/sample_input/hr_employees.csv @@ -0,0 +1,8 @@ +Name,Email,Department,Region,Hire Date,Salary +Laura Moss,laura@corp.com,HR,East,2021-03-15,55000 +Mike Stone,mike@corp.com,HR,West,2019-07-01,72000 +Nancy Hill,nancy@corp.com,Engineering,North,2022-11-20,98000 +Oscar Reed,oscar@corp.com,Engineering,South,2020-04-05,115000 +Paula Marsh,paula@corp.com,Marketing,East,2023-01-10,61000 +Quinn Ford,quinn@corp.com,Marketing,West,2018-09-30,68000 +Rachel Moore,rachel@corp.com,HR,North,2024-02-15,52000 diff --git a/Excel-CSV-File-Merger-main/sample_input/marketing_leads.csv b/Excel-CSV-File-Merger-main/sample_input/marketing_leads.csv new file mode 100644 index 0000000..524a299 --- /dev/null +++ b/Excel-CSV-File-Merger-main/sample_input/marketing_leads.csv @@ -0,0 +1,7 @@ + Name ,Email,Revenue, Region ,Date,Department +Sam Turner ,sam@leads.com,"$5,400.00", North ,2024/01/08,Marketing + Tina Brooks,tina@leads.com,8200,East,2024-02-14,Marketing +Uma Patel ,uma@leads.com,"$11,750.50",West ,"March 5, 2024",Marketing +Vince Hall,vince@leads.com,3100.0,South,2024-04-22,Marketing + Wendy Fox,wendy@leads.com,"$19,900",North,2024-05-30,Marketing +Xavier Long,xavier@leads.com,7650.25, East,2024-06-15,Marketing diff --git a/Excel-CSV-File-Merger-main/sample_input/ops_data.xlsx b/Excel-CSV-File-Merger-main/sample_input/ops_data.xlsx new file mode 100644 index 0000000..fe79f55 Binary files /dev/null and b/Excel-CSV-File-Merger-main/sample_input/ops_data.xlsx differ diff --git a/Excel-CSV-File-Merger-main/sample_input/sales_q1.xlsx b/Excel-CSV-File-Merger-main/sample_input/sales_q1.xlsx new file mode 100644 index 0000000..009705f Binary files /dev/null and b/Excel-CSV-File-Merger-main/sample_input/sales_q1.xlsx differ diff --git a/Excel-CSV-File-Merger-main/sample_input/sales_q2.xlsx b/Excel-CSV-File-Merger-main/sample_input/sales_q2.xlsx new file mode 100644 index 0000000..fb74ccd Binary files /dev/null and b/Excel-CSV-File-Merger-main/sample_input/sales_q2.xlsx differ diff --git a/Excel-CSV-File-Merger-main/tests/__pycache__/test_excel_merger.cpython-312-pytest-9.1.1.pyc b/Excel-CSV-File-Merger-main/tests/__pycache__/test_excel_merger.cpython-312-pytest-9.1.1.pyc new file mode 100644 index 0000000..11c4b21 Binary files /dev/null and b/Excel-CSV-File-Merger-main/tests/__pycache__/test_excel_merger.cpython-312-pytest-9.1.1.pyc differ diff --git a/Excel-CSV-File-Merger-main/tests/test_excel_merger.py b/Excel-CSV-File-Merger-main/tests/test_excel_merger.py new file mode 100644 index 0000000..a22756e --- /dev/null +++ b/Excel-CSV-File-Merger-main/tests/test_excel_merger.py @@ -0,0 +1,225 @@ +"""Tests for the Excel/CSV merge pipeline.""" +import os +import subprocess +import sys +from pathlib import Path + +import pandas as pd +import pytest + +PROJECT_ROOT = Path(__file__).resolve().parent.parent +sys.path.insert(0, str(PROJECT_ROOT)) + +from create_samples import create_samples # noqa: E402 +from excel_merger import ( # noqa: E402 + COLUMN_ALIASES, + clean_string, + dedupe_columns, + merge_files, + normalize_column_name, + normalize_currency, + normalize_dataframe, +) + + +class TestNormalizeCurrency: + @pytest.mark.parametrize("raw,expected", [ + ("$5,400.00", 5400.0), + ("5400", 5400.0), + (" $1,299.50 ", 1299.5), + ("-250.75", -250.75), + (3100.0, 3100.0), + ]) + def test_parses_currency_forms(self, raw, expected): + assert normalize_currency(raw) == expected + + @pytest.mark.parametrize("raw", ["", "N/A", None, float("nan")]) + def test_unparseable_returns_none(self, raw): + assert normalize_currency(raw) is None + + def test_rejects_series_with_clear_message(self): + """Guards the duplicate-column failure mode explicitly.""" + with pytest.raises(TypeError, match="duplicate column names"): + normalize_currency(pd.Series([1, 2])) + + +class TestNormalizeColumnName: + @pytest.mark.parametrize("raw,expected", [ + ("Sales Amount", "revenue"), + (" TERRITORY ", "region"), + ("Transaction Date", "date"), + ("Dept", "department"), + ("Unmapped Column", "unmapped column"), + ]) + def test_aliases_and_normalization(self, raw, expected): + assert normalize_column_name(raw) == expected + + def test_alias_map_is_idempotent(self): + """Canonical names must map to themselves, or merging is unstable.""" + for canonical in set(COLUMN_ALIASES.values()): + assert normalize_column_name(canonical) == canonical + + +class TestDedupeColumns: + def test_disambiguates_aliased_collisions(self): + # "Amount" and "Total" both alias to "revenue" + assert dedupe_columns(["name", "revenue", "revenue"], "f.csv") == [ + "name", "revenue", "revenue_2", + ] + + def test_leaves_unique_columns_untouched(self): + cols = ["name", "email", "revenue"] + assert dedupe_columns(cols, "f.csv") == cols + + def test_handles_triple_collision(self): + assert dedupe_columns(["r", "r", "r"], "f.csv") == ["r", "r_2", "r_3"] + + def test_generated_name_cannot_collide_with_an_existing_name(self): + cols = dedupe_columns(["x", "x", "x_2"], "f.csv") + assert cols == ["x", "x_2", "x_2_2"] + assert len(cols) == len(set(cols)) + + +class TestNormalizeDataframe: + def test_collision_does_not_raise(self): + """Regression: previously raised 'truth value of a Series is ambiguous'.""" + df = pd.DataFrame({"Name": ["a"], "Amount": ["$10"], "Total": ["$20"]}) + out = normalize_dataframe(df, "collide.csv") + assert out["revenue"].iloc[0] == 10.0 + assert "revenue_2" in out.columns + + def test_strips_whitespace_and_tags_source(self): + df = pd.DataFrame({"Name": [" Sam Turner "], "Amount": ["$5,400.00"]}) + out = normalize_dataframe(df, "leads.csv") + assert out["name"].iloc[0] == "Sam Turner" + assert out["revenue"].iloc[0] == 5400.0 + assert out["source_file"].iloc[0] == "leads.csv" + + def test_drops_fully_empty_rows(self): + df = pd.DataFrame({"Name": ["a", None], "Amount": ["$1", None]}) + assert len(normalize_dataframe(df, "f.csv")) == 1 + + def test_drops_rows_that_are_empty_after_trimming(self): + df = pd.DataFrame({ + "Name": [" ", None, " real "], + "Email": [None, "\t", ""], + }) + + out = normalize_dataframe(df, "f.csv") + + assert len(out) == 1 + assert out["name"].iloc[0] == "real" + assert out["email"].iloc[0] == "" + + def test_preserves_an_input_source_file_column(self): + df = pd.DataFrame({"source_file": ["user value"], "Name": ["a"]}) + out = normalize_dataframe(df, "actual.csv") + + assert out["source_file_2"].iloc[0] == "user value" + assert out["source_file"].iloc[0] == "actual.csv" + assert list(df.columns) == ["source_file", "Name"] + + +class TestMergeFiles: + def test_merges_dedups_and_writes_both_sheets(self, tmp_path): + src = tmp_path / "in" + src.mkdir() + pd.DataFrame({"Name": ["a", "b"], "Sales Amount": ["$1", "$2"]}).to_csv( + src / "one.csv", index=False) + pd.DataFrame({"name": ["b", "c"], "amount": ["$2", "$3"]}).to_csv( + src / "two.csv", index=False) + + out = tmp_path / "master.xlsx" + merge_files(src, out, deduplicate=True) + + assert out.exists() + sheets = pd.read_excel(out, sheet_name=None) + assert set(sheets) == {"Merged Data", "Merge Summary"} + merged = sheets["Merged Data"] + assert len(merged) == 3 # b/$2 deduplicated across files + assert list(merged.columns)[-1] == "source_file" + + def test_no_dedup_keeps_duplicates(self, tmp_path): + src = tmp_path / "in" + src.mkdir() + for n in ("one.csv", "two.csv"): + pd.DataFrame({"Name": ["a"], "Amount": ["$1"]}).to_csv(src / n, index=False) + out = tmp_path / "m.xlsx" + merge_files(src, out, deduplicate=False) + assert len(pd.read_excel(out, sheet_name="Merged Data")) == 2 + + def test_empty_folder_raises(self, tmp_path): + empty = tmp_path / "empty" + empty.mkdir() + with pytest.raises(FileNotFoundError): + merge_files(empty, tmp_path / "o.xlsx") + + def test_finds_nested_files_and_tracks_relative_source(self, tmp_path): + src = tmp_path / "in" + nested = src / "north" + nested.mkdir(parents=True) + pd.DataFrame({"Name": ["a"]}).to_csv(nested / "DATA.CSV", index=False) + + out = tmp_path / "master.xlsx" + merge_files(src, out) + + merged = pd.read_excel(out, sheet_name="Merged Data") + assert merged["source_file"].tolist() == ["north/DATA.CSV"] + + def test_rerun_does_not_ingest_its_own_output(self, tmp_path): + src = tmp_path / "in" + src.mkdir() + pd.DataFrame({"Name": ["a"]}).to_csv(src / "data.csv", index=False) + out = src / "master.xlsx" + + merge_files(src, out) + merge_files(src, out) + + summary = pd.read_excel(out, sheet_name="Merge Summary") + assert "master.xlsx" not in summary["Source File"].astype(str).tolist() + assert len(pd.read_excel(out, sheet_name="Merged Data")) == 1 + + def test_creates_missing_output_directories(self, tmp_path): + src = tmp_path / "in" + src.mkdir() + pd.DataFrame({"Name": ["a"]}).to_csv(src / "data.csv", index=False) + out = tmp_path / "new" / "nested" / "master.xlsx" + + merge_files(src, out) + + assert out.exists() + + +def test_clean_string_passes_through_non_strings(): + assert clean_string(42) == 42 + assert clean_string(" x ") == "x" + + +def test_import_has_no_log_file_side_effect(tmp_path): + env = os.environ.copy() + env["PYTHONPATH"] = str(PROJECT_ROOT) + + subprocess.run( + [sys.executable, "-W", "error", "-c", "import excel_merger"], + cwd=tmp_path, + env=env, + check=True, + capture_output=True, + text=True, + ) + + assert not (tmp_path / "merger.log").exists() + + +def test_sample_generator_creates_a_missing_destination(tmp_path): + destination = tmp_path / "new" / "sample_input" + + create_samples(destination) + + assert {path.name for path in destination.iterdir()} == { + "hr_employees.csv", + "marketing_leads.csv", + "ops_data.xlsx", + "sales_q1.xlsx", + "sales_q2.xlsx", + }