Name: Csv Handler
Availability: InStock
Author: datadrivenconstruction

CSV Handler for Construction Data

Overview

CSV is the universal exchange format in construction - from scheduling exports to cost databases. This skill handles encoding issues, delimiter detection, and data cleaning.

Python Implementation

import pandas as pd
import csv
from typing import Dict, Any, List, Optional, Tuple
from pathlib import Path
from dataclasses import dataclass
import chardet
@dataclass
class CSVProfile:
"""Profile of CSV file."""
encoding: str
delimiter: str
has_header: bool
row_count: int
column_count: int
columns: List[str]
class ConstructionCSVHandler:
"""Handle CSV files from construction software."""
COMMON_DELIMITERS = [&#x27;,&#x27;, &#x27;;&#x27;, &#x27;\t&#x27;, &#x27;|&#x27;]
COMMON_ENCODINGS = [&#x27;utf-8&#x27;, &#x27;utf-8-sig&#x27;, &#x27;latin-1&#x27;, &#x27;cp1252&#x27;, &#x27;iso-8859-1&#x27;]

def __init__(self):
    self.last_profile: Optional[CSVProfile] = None

def detect_encoding(self, file_path: str) -&gt; str:
    &quot;&quot;&quot;Detect file encoding.&quot;&quot;&quot;
    with open(file_path, &#x27;rb&#x27;) as f:
        raw = f.read(10000)
    result = chardet.detect(raw)
    return result.get(&#x27;encoding&#x27;, &#x27;utf-8&#x27;) or &#x27;utf-8&#x27;

def detect_delimiter(self, file_path: str, encoding: str) -&gt; str:
    &quot;&quot;&quot;Detect CSV delimiter.&quot;&quot;&quot;
    with open(file_path, &#x27;r&#x27;, encoding=encoding, errors=&#x27;replace&#x27;) as f:
        sample = f.read(5000)

    # Count occurrences
    counts = {d: sample.count(d) for d in self.COMMON_DELIMITERS}

    # Return most common that appears consistently
    if counts:
        return max(counts, key=counts.get)
    return &#x27;,&#x27;

def profile_csv(self, file_path: str) -&gt; CSVProfile:
    &quot;&quot;&quot;Profile CSV file.&quot;&quot;&quot;
    encoding = self.detect_encoding(file_path)
    delimiter = self.detect_delimiter(file_path, encoding)

    # Read sample
    df = pd.read_csv(file_path, encoding=encoding, delimiter=delimiter,
                     nrows=10, on_bad_lines=&#x27;skip&#x27;)

    has_header = not df.columns[0].replace(&#x27;.&#x27;, &#x27;&#x27;).replace(&#x27;-&#x27;, &#x27;&#x27;).isdigit()

    # Full row count
    with open(file_path, &#x27;r&#x27;, encoding=encoding, errors=&#x27;replace&#x27;) as f:
        row_count = sum(1 for _ in f) - (1 if has_header else 0)

    profile = CSVProfile(
        encoding=encoding,
        delimiter=delimiter,
        has_header=has_header,
        row_count=row_count,
        column_count=len(df.columns),
        columns=list(df.columns)
    )
    self.last_profile = profile
    return profile

def read_csv(self, file_path: str,
             encoding: Optional[str] = None,
             delimiter: Optional[str] = None,
             clean: bool = True) -&gt; pd.DataFrame:
    &quot;&quot;&quot;Read CSV with auto-detection.&quot;&quot;&quot;

    # Auto-detect if not provided
    if encoding is None:
        encoding = self.detect_encoding(file_path)
    if delimiter is None:
        delimiter = self.detect_delimiter(file_path, encoding)

    # Read with error handling
    df = pd.read_csv(
        file_path,
        encoding=encoding,
        delimiter=delimiter,
        on_bad_lines=&#x27;skip&#x27;,
        low_memory=False
    )

    if clean:
        df = self.clean_dataframe(df)

    return df

def clean_dataframe(self, df: pd.DataFrame) -&gt; pd.DataFrame:
    &quot;&quot;&quot;Clean construction CSV data.&quot;&quot;&quot;
    # Clean column names
    df.columns = [self._clean_column_name(c) for c in df.columns]

    # Remove empty rows and columns
    df = df.dropna(how=&#x27;all&#x27;)
    df = df.dropna(axis=1, how=&#x27;all&#x27;)

    # Strip whitespace from strings
    for col in df.select_dtypes(include=[&#x27;object&#x27;]):
        df[col] = df[col].str.strip() if df[col].dtype == &#x27;object&#x27; else df[col]

    return df

def _clean_column_name(self, name: str) -&gt; str:
    &quot;&quot;&quot;Clean column name.&quot;&quot;&quot;
    if not isinstance(name, str):
        return str(name)

    # Remove special characters, replace spaces
    clean = name.strip().lower()
    clean = clean.replace(&#x27; &#x27;, &#x27;_&#x27;).replace(&#x27;-&#x27;, &#x27;_&#x27;)
    clean = &#x27;&#x27;.join(c for c in clean if c.isalnum() or c == &#x27;_&#x27;)
    return clean

def merge_csvs(self, file_paths: List[str],
               on_column: Optional[str] = None) -&gt; pd.DataFrame:
    &quot;&quot;&quot;Merge multiple CSV files.&quot;&quot;&quot;
    dfs = []
    for path in file_paths:
        df = self.read_csv(path)
        df[&#x27;_source_file&#x27;] = Path(path).name
        dfs.append(df)

    if not dfs:
        return pd.DataFrame()

    if on_column and on_column in dfs[0].columns:
        result = dfs[0]
        for df in dfs[1:]:
            result = pd.merge(result, df, on=on_column, how=&#x27;outer&#x27;)
        return result

    return pd.concat(dfs, ignore_index=True)

def split_csv(self, df: pd.DataFrame,
              group_column: str,
              output_dir: str) -&gt; List[str]:
    &quot;&quot;&quot;Split CSV by column values.&quot;&quot;&quot;
    output_path = Path(output_dir)
    output_path.mkdir(parents=True, exist_ok=True)

    files = []
    for value in df[group_column].unique():
        subset = df[df[group_column] == value]
        filename = f&quot;{group_column}_{value}.csv&quot;
        filepath = output_path / filename
        subset.to_csv(filepath, index=False)
        files.append(str(filepath))

    return files

def convert_types(self, df: pd.DataFrame,
                  type_map: Dict[str, str] = None) -&gt; pd.DataFrame:
    &quot;&quot;&quot;Convert column types intelligently.&quot;&quot;&quot;
    df = df.copy()

    if type_map:
        for col, dtype in type_map.items():
            if col in df.columns:
                try:
                    df[col] = df[col].astype(dtype)
                except:
                    pass
    else:
        # Auto-convert
        for col in df.columns:
            # Try numeric
            try:
                df[col] = pd.to_numeric(df[col])
                continue
            except:
                pass

            # Try datetime
            try:
                df[col] = pd.to_datetime(df[col])
            except:
                pass

    return df

def export_csv(self, df: pd.DataFrame,
               file_path: str,
               encoding: str = &#x27;utf-8-sig&#x27;,
               delimiter: str = &#x27;,&#x27;) -&gt; str:
    &quot;&quot;&quot;Export DataFrame to CSV.&quot;&quot;&quot;
    df.to_csv(file_path, encoding=encoding, sep=delimiter, index=False)
    return file_path

Specialized handlers
class ScheduleCSVHandler(ConstructionCSVHandler):
"""Handler for project schedule CSVs."""
SCHEDULE_COLUMNS = [&#x27;task_id&#x27;, &#x27;task_name&#x27;, &#x27;start_date&#x27;, &#x27;end_date&#x27;,
                    &#x27;duration&#x27;, &#x27;predecessors&#x27;, &#x27;resources&#x27;]

def parse_schedule(self, file_path: str) -&gt; pd.DataFrame:
    &quot;&quot;&quot;Parse schedule CSV.&quot;&quot;&quot;
    df = self.read_csv(file_path)

    # Convert date columns
    for col in df.columns:
        if &#x27;date&#x27; in col.lower() or &#x27;start&#x27; in col.lower() or &#x27;end&#x27; in col.lower():
            try:
                df[col] = pd.to_datetime(df[col])
            except:
                pass

    return df

class CostCSVHandler(ConstructionCSVHandler):
"""Handler for cost/estimate CSVs."""
def parse_costs(self, file_path: str) -&gt; pd.DataFrame:
    &quot;&quot;&quot;Parse cost CSV.&quot;&quot;&quot;
    df = self.read_csv(file_path)

    # Find and convert numeric columns
    for col in df.columns:
        if any(word in col.lower() for word in [&#x27;cost&#x27;, &#x27;price&#x27;, &#x27;amount&#x27;, &#x27;total&#x27;, &#x27;qty&#x27;, &#x27;quantity&#x27;]):
            df[col] = pd.to_numeric(df[col].replace(r&#x27;[\$,]&#x27;, &#x27;&#x27;, regex=True), errors=&#x27;coerce&#x27;)

    return df

Quick Start

handler = ConstructionCSVHandler()
Profile CSV first
profile = handler.profile_csv("export.csv")
print(f"Encoding: {profile.encoding}, Delimiter: '{profile.delimiter}'")
Read with auto-detection
df = handler.read_csv("export.csv")
print(f"Loaded {len(df)} rows, {len(df.columns)} columns")

Common Use Cases

1. Merge Multiple Exports

files = ["jan_export.csv", "feb_export.csv", "mar_export.csv"]
merged = handler.merge_csvs(files)

2. Split by Category

handler.split_csv(df, group_column='category', output_dir='./split_files')

3. Schedule Import

schedule_handler = ScheduleCSVHandler()
schedule = schedule_handler.parse_schedule("p6_export.csv")

Resources

DDC Book: Chapter 2.1 - Structured Data

Csv Handler

AI Skill Market Insights

Be Part of the 2,288+ Developer Community

CSV Handler for Construction Data

Overview

Python Implementation

Specialized handlers

Quick Start

Profile CSV first

Read with auto-detection

Common Use Cases

1. Merge Multiple Exports

2. Split by Category

3. Schedule Import

Resources

Quick Start

Manual Installation

TEAR & SHARE

Tags

Data Engineer

Data Scientist

Data Analysis

PostgreSQL

Snowflake MCP Connection

Channels

Learn

Compare

Company

Agents