From ed388fe59f91e91c643f3d9a398631a2b02cf841 Mon Sep 17 00:00:00 2001 From: Vishesh 'ironeagle' Bangotra Date: Mon, 24 Aug 2026 14:43:51 +0530 Subject: [PATCH] add csv support: ContentType.CSV + stdlib-csv-backed client/scraper/parser --- omniread/__init__.py | 14 +++++ omniread/core/content.py | 3 ++ omniread/csv/__init__.py | 26 +++++++++ omniread/csv/client.py | 97 ++++++++++++++++++++++++++++++++++ omniread/csv/parser.py | 102 ++++++++++++++++++++++++++++++++++++ omniread/csv/parser_base.py | 55 +++++++++++++++++++ omniread/csv/scraper.py | 79 ++++++++++++++++++++++++++++ 7 files changed, 376 insertions(+) create mode 100644 omniread/csv/__init__.py create mode 100644 omniread/csv/client.py create mode 100644 omniread/csv/parser.py create mode 100644 omniread/csv/parser_base.py create mode 100644 omniread/csv/scraper.py diff --git a/omniread/__init__.py b/omniread/__init__.py index 196bab0..46f53da 100644 --- a/omniread/__init__.py +++ b/omniread/__init__.py @@ -111,6 +111,13 @@ required. """ from .core import Content, ContentType +from .csv import ( + BaseCsvClient, + CsvParser, + CsvParserBase, + CsvScraper, + FileSystemCsvClient, +) from .html import HTMLScraper, HTMLParser from .pdf import FileSystemPDFClient, PDFScraper, PDFParser from .xlsx import ( @@ -135,6 +142,13 @@ __all__ = [ "PDFScraper", "PDFParser", + # csv + "BaseCsvClient", + "FileSystemCsvClient", + "CsvScraper", + "CsvParser", + "CsvParserBase", + # xlsx "BaseXlsxClient", "FileSystemXlsxClient", diff --git a/omniread/core/content.py b/omniread/core/content.py index 65a1456..775a843 100644 --- a/omniread/core/content.py +++ b/omniread/core/content.py @@ -38,6 +38,9 @@ class ContentType(str, Enum): XLSX = "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" """Office Open XML spreadsheet (xlsx/xlsm) content.""" + CSV = "text/csv" + """Comma-separated-value document content.""" + JSON = "application/json" """JSON document content.""" diff --git a/omniread/csv/__init__.py b/omniread/csv/__init__.py new file mode 100644 index 0000000..ae704d2 --- /dev/null +++ b/omniread/csv/__init__.py @@ -0,0 +1,26 @@ +""" +# Summary + +CSV subpackage for OmniRead. + +Provides acquisition and parsing of comma-separated-value content: + +- `BaseCsvClient`: abstract backing-store client for csv bytes. +- `FileSystemCsvClient`: local filesystem implementation. +- `CsvScraper`: wraps fetched bytes into canonical `Content`. +- `CsvParserBase`: content-type-enforcing parser contract. +- `CsvParser`: generic string-row parser built on the standard csv module. +""" + +from .client import BaseCsvClient, FileSystemCsvClient +from .parser import CsvParser +from .parser_base import CsvParserBase +from .scraper import CsvScraper + +__all__ = [ + "BaseCsvClient", + "FileSystemCsvClient", + "CsvScraper", + "CsvParser", + "CsvParserBase", +] diff --git a/omniread/csv/client.py b/omniread/csv/client.py new file mode 100644 index 0000000..22ad75f --- /dev/null +++ b/omniread/csv/client.py @@ -0,0 +1,97 @@ +""" +# Summary + +CSV client abstractions for OmniRead. + +This module defines the **client layer** responsible for retrieving raw +comma-separated-value document bytes from a concrete backing store. + +Clients provide low-level access to csv binaries and are intentionally +decoupled from scraping and parsing logic. They do not perform validation, +interpretation, or content extraction. + +Typical backing stores include: + +- Local filesystems +- Object storage (S3, GCS, etc.) +- Network file systems +""" + +from typing import Any +from abc import ABC, abstractmethod +from pathlib import Path + + +class BaseCsvClient(ABC): + """ + Abstract client responsible for retrieving csv bytes. + + Retrieves bytes from a specific backing store (filesystem, S3, FTP, etc.). + + Notes: + **Responsibilities:** + + - Implementations must accept a source identifier appropriate to + the backing store. + - Return the full csv binary payload. + - Raise retrieval-specific errors on failure. + """ + + @abstractmethod + def fetch(self, source: Any) -> bytes: + """ + Fetch raw csv bytes from the given source. + + Args: + source (Any): + Identifier of the csv location, such as a file path, + object storage key, or remote reference. + + Returns: + bytes: + Raw csv bytes. + + Raises: + Exception: + Retrieval-specific errors defined by the implementation. + """ + raise NotImplementedError + + +class FileSystemCsvClient(BaseCsvClient): + """ + CSV client that reads from the local filesystem. + + Notes: + **Guarantees:** + + - This client reads csv files directly from the disk and + returns their raw binary contents. + """ + + def fetch(self, path: Path) -> bytes: + """ + Read a csv file from the local filesystem. + + Args: + path (Path): + Filesystem path to the csv file. + + Returns: + bytes: + Raw csv bytes. + + Raises: + FileNotFoundError: + If the path does not exist. + ValueError: + If the path exists but is not a file. + """ + + if not path.exists(): + raise FileNotFoundError(f"csv not found: {path}") + + if not path.is_file(): + raise ValueError(f"Path is not a file: {path}") + + return path.read_bytes() diff --git a/omniread/csv/parser.py b/omniread/csv/parser.py new file mode 100644 index 0000000..358e664 --- /dev/null +++ b/omniread/csv/parser.py @@ -0,0 +1,102 @@ +""" +# Summary + +CSV parser implementations for OmniRead. + +This module provides a concrete, generic parser for comma-separated-value +documents. It exposes records as lists of string cells so downstream +consumers can interpret tabular content without depending on the ``csv`` +module directly. + +The parser is intentionally statement-agnostic: it performs no header +detection or column interpretation beyond basic cell normalization and +delimiter detection. +""" + +from io import StringIO +from csv import Sniffer, reader +from typing import List + +from omniread.core.content import Content +from .parser_base import CsvParserBase + + +class CsvParser(CsvParserBase): + """ + Generic csv parser producing string rows from the document. + + Notes: + **Responsibilities:** + + - Decode the payload (UTF-8 with BOM support, Latin-1 fallback). + - Detect the delimiter from a leading sample (`,` `;` tab `|`), + defaulting to `,`. + - Normalize cells into deterministic stripped string values. + - Expose row extraction helpers mirroring `XlsxParser.rows`. + + **Constraints:** + + - All values are strings; consumers requiring typed values must + convert on their side. + - Quoted fields containing delimiters/newlines are handled by + the standard ``csv`` module. + """ + + _DELIMITERS = ",;\t|" + + def __init__(self, content: Content): + """ + Initialize the parser. + + Args: + content (Content): + CSV content to parse; its type must be supported. + """ + super().__init__(content) + + def parse(self) -> List[List[str]]: + """ + Parse the document into normalized string rows. + + Returns: + List[List[str]]: + Rows of the document. + """ + return self.rows() + + def rows(self, *, skip_empty: bool = True) -> List[List[str]]: + """ + Extract normalized string rows from the document. + + Args: + skip_empty (bool): + When True (default), rows whose cells are all blank are + omitted. + + Returns: + List[List[str]]: + Normalized rows; trailing blank cells are trimmed per row. + """ + text = self._decode(self.content.raw) + dialect_sample = text[:4096] + try: + dialect = Sniffer().sniff(dialect_sample, delimiters=self._DELIMITERS) + delimiter = dialect.delimiter + except Exception: + delimiter = "," + out: List[List[str]] = [] + for row in reader(StringIO(text), delimiter=delimiter): + cells = [c.strip() for c in row] + while cells and not cells[-1]: + cells.pop() + if skip_empty and not any(cells): + continue + out.append(cells) + return out + + @staticmethod + def _decode(raw: bytes) -> str: + try: + return raw.decode("utf-8-sig") + except UnicodeDecodeError: + return raw.decode("latin-1") diff --git a/omniread/csv/parser_base.py b/omniread/csv/parser_base.py new file mode 100644 index 0000000..5daf885 --- /dev/null +++ b/omniread/csv/parser_base.py @@ -0,0 +1,55 @@ +""" +# Summary + +CSV parser base implementation for OmniRead. + +This module defines the **CSV-specific parser contract**, extending the +format-agnostic `BaseParser` with constraints appropriate for +comma-separated-value documents. +""" + +from typing import Generic, TypeVar +from abc import abstractmethod + +from omniread.core.content import ContentType +from omniread.core.parser import BaseParser + +T = TypeVar("T") + + +class CsvParserBase(BaseParser[T], Generic[T]): + """ + Base csv parser. + + Notes: + **Responsibilities:** + + - This class enforces csv content-type compatibility and provides + the extension point for implementing concrete csv parsing + strategies. + + **Constraints:** + + - Concrete implementations must define the output type `T` and + implement the `parse()` method. + """ + + supported_types = {ContentType.CSV} + """ + Set of content types supported by this parser (CSV only). + """ + + @abstractmethod + def parse(self) -> T: + """ + Parse csv content into a structured output. + + Returns: + T: + Parsed representation of type `T`. + + Raises: + Exception: + Parsing-specific errors as defined by the implementation. + """ + raise NotImplementedError diff --git a/omniread/csv/scraper.py b/omniread/csv/scraper.py new file mode 100644 index 0000000..5785dfc --- /dev/null +++ b/omniread/csv/scraper.py @@ -0,0 +1,79 @@ +""" +# Summary + +CSV scraper for OmniRead. + +This module defines the scraper responsible for acquiring raw +comma-separated-value document content from a backing store via a +configured client. + +The scraper does not interpret or parse the acquired bytes; it wraps them in +the canonical `Content` model. +""" + +from typing import Any, Mapping, Optional + +from omniread.core.content import Content, ContentType +from .client import BaseCsvClient + + +class CsvScraper: + """ + Scraper for csv documents. + + Notes: + **Responsibilities:** + + - Fetch raw csv bytes via the configured client. + - Wrap the payload in a canonical `Content` instance with the + CSV content type and source identifier. + + **Constraints:** + + - The scraper does not perform parsing or interpretation. + - Does not assume a specific storage backend. + """ + + def __init__(self, *, client: BaseCsvClient): + """ + Initialize the CSV scraper. + + Args: + client (BaseCsvClient): + Client responsible for retrieving raw csv bytes. + """ + self._client = client + + def fetch( + self, + source: Any, + *, + metadata: Optional[Mapping[str, Any]] = None, + ) -> Content: + """ + Fetch a csv document from the given source. + + Args: + source (Any): + Identifier of the csv source as understood by the + configured client. + metadata (Optional[Mapping[str, Any]], optional): + Optional metadata to attach to the returned content. + + Returns: + Content: + A `Content` instance containing raw csv bytes, source + identifier, CSV content type, and optional metadata. + + Raises: + Exception: + Retrieval-specific errors raised by the client. + """ + raw = self._client.fetch(source) + + return Content( + raw=raw, + source=source, + content_type=ContentType.CSV, + metadata=dict(metadata) if metadata else None, + )