From aedd2061e17621e34d99c069dc861da345b8cb60 Mon Sep 17 00:00:00 2001 From: Vishesh 'ironeagle' Bangotra Date: Wed, 16 Sep 2026 19:59:00 +0530 Subject: [PATCH] feat: collect omniread wiki; register wiki kind; refresh lib and mcp artifacts Adds the omniread wiki to the hub alongside lib and mcp and completes the library nav with csv/xlsx groups, picking up regenerated MCP modules. --- _index/index.html | 1 + config.yml | 1 + mcp/omniread/index.json | 2 +- .../modules/omniread.core.content.json | 32 +- mcp/omniread/modules/omniread.core.json | 170 +- .../modules/omniread.core.parser.json | 58 +- .../modules/omniread.core.scraper.json | 42 +- mcp/omniread/modules/omniread.csv.client.json | 36 +- mcp/omniread/modules/omniread.csv.json | 198 +- mcp/omniread/modules/omniread.csv.parser.json | 43 +- .../modules/omniread.csv.parser_base.json | 49 +- .../modules/omniread.csv.scraper.json | 46 +- mcp/omniread/modules/omniread.html.json | 165 +- .../modules/omniread.html.parser.json | 88 +- .../modules/omniread.html.scraper.json | 55 +- mcp/omniread/modules/omniread.json | 1003 +-- mcp/omniread/modules/omniread.pdf.client.json | 36 +- mcp/omniread/modules/omniread.pdf.json | 149 +- mcp/omniread/modules/omniread.pdf.parser.json | 49 +- .../modules/omniread.pdf.scraper.json | 50 +- .../modules/omniread.xlsx.client.json | 36 +- mcp/omniread/modules/omniread.xlsx.json | 209 +- .../modules/omniread.xlsx.parser.json | 50 +- .../modules/omniread.xlsx.parser_base.json | 49 +- .../modules/omniread.xlsx.scraper.json | 46 +- omniread/lib/404.html | 334 + omniread/lib/core/content/index.html | 334 + omniread/lib/core/index.html | 338 +- omniread/lib/core/parser/index.html | 336 +- omniread/lib/core/scraper/index.html | 336 +- omniread/lib/csv/client/index.html | 499 +- omniread/lib/csv/index.html | 352 +- omniread/lib/csv/parser/index.html | 504 +- omniread/lib/csv/parser_base/index.html | 495 +- omniread/lib/csv/scraper/index.html | 462 +- omniread/lib/html/index.html | 342 +- omniread/lib/html/parser/index.html | 338 +- omniread/lib/html/scraper/index.html | 338 +- omniread/lib/index.html | 350 +- omniread/lib/objects.inv | Bin 1419 -> 1286 bytes omniread/lib/omniread/core/content/index.html | 1333 ---- omniread/lib/omniread/core/index.html | 1921 ----- omniread/lib/omniread/core/parser/index.html | 1167 --- omniread/lib/omniread/core/scraper/index.html | 1069 --- omniread/lib/omniread/csv/client/index.html | 1216 --- omniread/lib/omniread/csv/index.html | 2135 ------ omniread/lib/omniread/csv/parser/index.html | 1188 --- .../lib/omniread/csv/parser_base/index.html | 1143 --- omniread/lib/omniread/csv/scraper/index.html | 1070 --- omniread/lib/omniread/html/index.html | 1904 ----- omniread/lib/omniread/html/parser/index.html | 1490 ---- omniread/lib/omniread/html/scraper/index.html | 1228 --- omniread/lib/omniread/index.html | 3294 -------- omniread/lib/omniread/pdf/client/index.html | 1215 --- omniread/lib/omniread/pdf/index.html | 1612 ---- omniread/lib/omniread/pdf/parser/index.html | 1151 --- omniread/lib/omniread/pdf/scraper/index.html | 1069 --- omniread/lib/omniread/xlsx/client/index.html | 1216 --- omniread/lib/omniread/xlsx/index.html | 2279 ------ omniread/lib/omniread/xlsx/parser/index.html | 1330 ---- .../lib/omniread/xlsx/parser_base/index.html | 1143 --- omniread/lib/omniread/xlsx/scraper/index.html | 1069 --- omniread/lib/pdf/client/index.html | 334 + omniread/lib/pdf/index.html | 342 +- omniread/lib/pdf/parser/index.html | 338 +- omniread/lib/pdf/scraper/index.html | 340 +- omniread/lib/search/search_index.json | 2 +- omniread/lib/sitemap.xml.gz | Bin 127 -> 127 bytes omniread/lib/xlsx/client/index.html | 497 ++ omniread/lib/xlsx/index.html | 346 +- omniread/lib/xlsx/parser/index.html | 520 +- omniread/lib/xlsx/parser_base/index.html | 495 +- omniread/lib/xlsx/scraper/index.html | 458 +- omniread/wiki/01_overview/index.html | 751 ++ omniread/wiki/02_how_to_use/index.html | 724 ++ omniread/wiki/03_extending/index.html | 714 ++ omniread/wiki/04_development/index.html | 696 ++ omniread/wiki/404.html | 478 ++ omniread/wiki/assets/images/favicon.png | Bin 0 -> 1870 bytes .../assets/javascripts/bundle.f55a23d4.min.js | 16 + .../javascripts/bundle.f55a23d4.min.js.map | 7 + .../javascripts/lunr/min/lunr.ar.min.js | 1 + .../javascripts/lunr/min/lunr.da.min.js | 18 + .../javascripts/lunr/min/lunr.de.min.js | 18 + .../javascripts/lunr/min/lunr.du.min.js | 18 + .../javascripts/lunr/min/lunr.el.min.js | 1 + .../javascripts/lunr/min/lunr.es.min.js | 18 + .../javascripts/lunr/min/lunr.fi.min.js | 18 + .../javascripts/lunr/min/lunr.fr.min.js | 18 + .../javascripts/lunr/min/lunr.he.min.js | 1 + .../javascripts/lunr/min/lunr.hi.min.js | 1 + .../javascripts/lunr/min/lunr.hu.min.js | 18 + .../javascripts/lunr/min/lunr.hy.min.js | 1 + .../javascripts/lunr/min/lunr.it.min.js | 18 + .../javascripts/lunr/min/lunr.ja.min.js | 1 + .../javascripts/lunr/min/lunr.jp.min.js | 1 + .../javascripts/lunr/min/lunr.kn.min.js | 1 + .../javascripts/lunr/min/lunr.ko.min.js | 1 + .../javascripts/lunr/min/lunr.multi.min.js | 1 + .../javascripts/lunr/min/lunr.nl.min.js | 18 + .../javascripts/lunr/min/lunr.no.min.js | 18 + .../javascripts/lunr/min/lunr.pt.min.js | 18 + .../javascripts/lunr/min/lunr.ro.min.js | 18 + .../javascripts/lunr/min/lunr.ru.min.js | 18 + .../javascripts/lunr/min/lunr.sa.min.js | 1 + .../lunr/min/lunr.stemmer.support.min.js | 1 + .../javascripts/lunr/min/lunr.sv.min.js | 18 + .../javascripts/lunr/min/lunr.ta.min.js | 1 + .../javascripts/lunr/min/lunr.te.min.js | 1 + .../javascripts/lunr/min/lunr.th.min.js | 1 + .../javascripts/lunr/min/lunr.tr.min.js | 18 + .../javascripts/lunr/min/lunr.vi.min.js | 1 + .../javascripts/lunr/min/lunr.zh.min.js | 1 + .../wiki/assets/javascripts/lunr/tinyseg.js | 206 + .../wiki/assets/javascripts/lunr/wordcut.js | 6708 +++++++++++++++++ .../workers/search.973d3a69.min.js | 42 + .../workers/search.973d3a69.min.js.map | 7 + .../assets/stylesheets/main.84d31ad4.min.css | 1 + .../stylesheets/main.84d31ad4.min.css.map | 1 + .../stylesheets/palette.06af60db.min.css | 1 + .../stylesheets/palette.06af60db.min.css.map | 1 + omniread/wiki/index.html | 661 ++ omniread/wiki/search/search_index.json | 1 + omniread/wiki/sitemap.xml | 3 + omniread/wiki/sitemap.xml.gz | Bin 0 -> 127 bytes 125 files changed, 21039 insertions(+), 34201 deletions(-) delete mode 100644 omniread/lib/omniread/core/content/index.html delete mode 100644 omniread/lib/omniread/core/index.html delete mode 100644 omniread/lib/omniread/core/parser/index.html delete mode 100644 omniread/lib/omniread/core/scraper/index.html delete mode 100644 omniread/lib/omniread/csv/client/index.html delete mode 100644 omniread/lib/omniread/csv/index.html delete mode 100644 omniread/lib/omniread/csv/parser/index.html delete mode 100644 omniread/lib/omniread/csv/parser_base/index.html delete mode 100644 omniread/lib/omniread/csv/scraper/index.html delete mode 100644 omniread/lib/omniread/html/index.html delete mode 100644 omniread/lib/omniread/html/parser/index.html delete mode 100644 omniread/lib/omniread/html/scraper/index.html delete mode 100644 omniread/lib/omniread/index.html delete mode 100644 omniread/lib/omniread/pdf/client/index.html delete mode 100644 omniread/lib/omniread/pdf/index.html delete mode 100644 omniread/lib/omniread/pdf/parser/index.html delete mode 100644 omniread/lib/omniread/pdf/scraper/index.html delete mode 100644 omniread/lib/omniread/xlsx/client/index.html delete mode 100644 omniread/lib/omniread/xlsx/index.html delete mode 100644 omniread/lib/omniread/xlsx/parser/index.html delete mode 100644 omniread/lib/omniread/xlsx/parser_base/index.html delete mode 100644 omniread/lib/omniread/xlsx/scraper/index.html create mode 100644 omniread/wiki/01_overview/index.html create mode 100644 omniread/wiki/02_how_to_use/index.html create mode 100644 omniread/wiki/03_extending/index.html create mode 100644 omniread/wiki/04_development/index.html create mode 100644 omniread/wiki/404.html create mode 100644 omniread/wiki/assets/images/favicon.png create mode 100644 omniread/wiki/assets/javascripts/bundle.f55a23d4.min.js create mode 100644 omniread/wiki/assets/javascripts/bundle.f55a23d4.min.js.map create mode 100644 omniread/wiki/assets/javascripts/lunr/min/lunr.ar.min.js create mode 100644 omniread/wiki/assets/javascripts/lunr/min/lunr.da.min.js create mode 100644 omniread/wiki/assets/javascripts/lunr/min/lunr.de.min.js create mode 100644 omniread/wiki/assets/javascripts/lunr/min/lunr.du.min.js create mode 100644 omniread/wiki/assets/javascripts/lunr/min/lunr.el.min.js create mode 100644 omniread/wiki/assets/javascripts/lunr/min/lunr.es.min.js create mode 100644 omniread/wiki/assets/javascripts/lunr/min/lunr.fi.min.js create mode 100644 omniread/wiki/assets/javascripts/lunr/min/lunr.fr.min.js create mode 100644 omniread/wiki/assets/javascripts/lunr/min/lunr.he.min.js create mode 100644 omniread/wiki/assets/javascripts/lunr/min/lunr.hi.min.js create mode 100644 omniread/wiki/assets/javascripts/lunr/min/lunr.hu.min.js create mode 100644 omniread/wiki/assets/javascripts/lunr/min/lunr.hy.min.js create mode 100644 omniread/wiki/assets/javascripts/lunr/min/lunr.it.min.js create mode 100644 omniread/wiki/assets/javascripts/lunr/min/lunr.ja.min.js create mode 100644 omniread/wiki/assets/javascripts/lunr/min/lunr.jp.min.js create mode 100644 omniread/wiki/assets/javascripts/lunr/min/lunr.kn.min.js create mode 100644 omniread/wiki/assets/javascripts/lunr/min/lunr.ko.min.js create mode 100644 omniread/wiki/assets/javascripts/lunr/min/lunr.multi.min.js create mode 100644 omniread/wiki/assets/javascripts/lunr/min/lunr.nl.min.js create mode 100644 omniread/wiki/assets/javascripts/lunr/min/lunr.no.min.js create mode 100644 omniread/wiki/assets/javascripts/lunr/min/lunr.pt.min.js create mode 100644 omniread/wiki/assets/javascripts/lunr/min/lunr.ro.min.js create mode 100644 omniread/wiki/assets/javascripts/lunr/min/lunr.ru.min.js create mode 100644 omniread/wiki/assets/javascripts/lunr/min/lunr.sa.min.js create mode 100644 omniread/wiki/assets/javascripts/lunr/min/lunr.stemmer.support.min.js create mode 100644 omniread/wiki/assets/javascripts/lunr/min/lunr.sv.min.js create mode 100644 omniread/wiki/assets/javascripts/lunr/min/lunr.ta.min.js create mode 100644 omniread/wiki/assets/javascripts/lunr/min/lunr.te.min.js create mode 100644 omniread/wiki/assets/javascripts/lunr/min/lunr.th.min.js create mode 100644 omniread/wiki/assets/javascripts/lunr/min/lunr.tr.min.js create mode 100644 omniread/wiki/assets/javascripts/lunr/min/lunr.vi.min.js create mode 100644 omniread/wiki/assets/javascripts/lunr/min/lunr.zh.min.js create mode 100644 omniread/wiki/assets/javascripts/lunr/tinyseg.js create mode 100644 omniread/wiki/assets/javascripts/lunr/wordcut.js create mode 100644 omniread/wiki/assets/javascripts/workers/search.973d3a69.min.js create mode 100644 omniread/wiki/assets/javascripts/workers/search.973d3a69.min.js.map create mode 100644 omniread/wiki/assets/stylesheets/main.84d31ad4.min.css create mode 100644 omniread/wiki/assets/stylesheets/main.84d31ad4.min.css.map create mode 100644 omniread/wiki/assets/stylesheets/palette.06af60db.min.css create mode 100644 omniread/wiki/assets/stylesheets/palette.06af60db.min.css.map create mode 100644 omniread/wiki/index.html create mode 100644 omniread/wiki/search/search_index.json create mode 100644 omniread/wiki/sitemap.xml create mode 100644 omniread/wiki/sitemap.xml.gz diff --git a/_index/index.html b/_index/index.html index b54550d..df1fa42 100644 --- a/_index/index.html +++ b/_index/index.html @@ -158,6 +158,7 @@

Omniread

Unified ingestion and normalization layer for structured and unstructured data sources.

+ view wiki view lib
diff --git a/config.yml b/config.yml index 32dec7c..d50dae6 100644 --- a/config.yml +++ b/config.yml @@ -66,6 +66,7 @@ repos: description: Unified ingestion and normalization layer for structured and unstructured data sources. section: libraries docs: + wiki: site lib: site mcp: { bundle: docs/mcp, server: omniread, port: 8003 } diff --git a/mcp/omniread/index.json b/mcp/omniread/index.json index 88b932a..c73df60 100644 --- a/mcp/omniread/index.json +++ b/mcp/omniread/index.json @@ -1,5 +1,5 @@ { - "project": "omniread", + "project": "OmniRead", "type": "docforge-model", "modules_count": 22, "source": "docforge" diff --git a/mcp/omniread/modules/omniread.core.content.json b/mcp/omniread/modules/omniread.core.content.json index b88d924..a71ec4f 100644 --- a/mcp/omniread/modules/omniread.core.content.json +++ b/mcp/omniread/modules/omniread.core.content.json @@ -4,39 +4,11 @@ "path": "omniread.core.content", "docstring": "# Summary\n\nCanonical content models for OmniRead.\n\nThis module defines the **format-agnostic content representation** used across\nall parsers and scrapers in OmniRead.\n\nThe models defined here represent *what* was extracted, not *how* it was\nretrieved or parsed. Format-specific behavior and metadata must not alter\nthe semantic meaning of these models.", "objects": { - "Mapping": { - "name": "Mapping", - "kind": "alias", - "path": "omniread.core.content.Mapping", - "signature": "", - "docstring": null - }, - "dataclass": { - "name": "dataclass", - "kind": "alias", - "path": "omniread.core.content.dataclass", - "signature": "", - "docstring": null - }, - "Enum": { - "name": "Enum", - "kind": "alias", - "path": "omniread.core.content.Enum", - "signature": "", - "docstring": null - }, - "Any": { - "name": "Any", - "kind": "alias", - "path": "omniread.core.content.Any", - "signature": "", - "docstring": null - }, "ContentType": { "name": "ContentType", "kind": "class", "path": "omniread.core.content.ContentType", - "signature": "", + "signature": null, "docstring": "Supported MIME types for extracted content.\n\nNotes:\n **Guarantees:**\n\n - This enum represents the declared or inferred media type of the\n content source.\n - It is primarily used for routing content to the appropriate\n parser or downstream consumer.", "members": { "HTML": { @@ -87,7 +59,7 @@ "name": "Content", "kind": "class", "path": "omniread.core.content.Content", - "signature": "", + "signature": "Content(raw: bytes, source: str, content_type: ContentType | None = ..., metadata: Mapping[str, Any] | None = ...)", "docstring": "Normalized representation of extracted content.\n\nNotes:\n **Responsibilities:**\n\n - A `Content` instance represents a raw content payload along with\n minimal contextual metadata describing its origin and type.\n - This class is the primary exchange format between scrapers,\n parsers, and downstream consumers.", "members": { "raw": { diff --git a/mcp/omniread/modules/omniread.core.json b/mcp/omniread/modules/omniread.core.json index 8d60ecf..fd51d72 100644 --- a/mcp/omniread/modules/omniread.core.json +++ b/mcp/omniread/modules/omniread.core.json @@ -8,35 +8,35 @@ "name": "Content", "kind": "class", "path": "omniread.core.Content", - "signature": "", + "signature": "Content(raw: bytes, source: str, content_type: ContentType | None = ..., metadata: Mapping[str, Any] | None = ...)", "docstring": "Normalized representation of extracted content.\n\nNotes:\n **Responsibilities:**\n\n - A `Content` instance represents a raw content payload along with\n minimal contextual metadata describing its origin and type.\n - This class is the primary exchange format between scrapers,\n parsers, and downstream consumers.", "members": { "raw": { "name": "raw", "kind": "attribute", "path": "omniread.core.Content.raw", - "signature": "", + "signature": null, "docstring": "Raw content bytes as retrieved from the source." }, "source": { "name": "source", "kind": "attribute", "path": "omniread.core.Content.source", - "signature": "", + "signature": null, "docstring": "Identifier of the content origin (URL, file path, or logical name)." }, "content_type": { "name": "content_type", "kind": "attribute", "path": "omniread.core.Content.content_type", - "signature": "", + "signature": null, "docstring": "Optional MIME type of the content, if known." }, "metadata": { "name": "metadata", "kind": "attribute", "path": "omniread.core.Content.metadata", - "signature": "", + "signature": null, "docstring": "Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes)." } } @@ -45,49 +45,49 @@ "name": "ContentType", "kind": "class", "path": "omniread.core.ContentType", - "signature": "", + "signature": null, "docstring": "Supported MIME types for extracted content.\n\nNotes:\n **Guarantees:**\n\n - This enum represents the declared or inferred media type of the\n content source.\n - It is primarily used for routing content to the appropriate\n parser or downstream consumer.", "members": { "HTML": { "name": "HTML", "kind": "attribute", "path": "omniread.core.ContentType.HTML", - "signature": "", + "signature": null, "docstring": "HTML document content." }, "PDF": { "name": "PDF", "kind": "attribute", "path": "omniread.core.ContentType.PDF", - "signature": "", + "signature": null, "docstring": "PDF document content." }, "XLSX": { "name": "XLSX", "kind": "attribute", "path": "omniread.core.ContentType.XLSX", - "signature": "", + "signature": null, "docstring": "Office Open XML spreadsheet (xlsx/xlsm) content." }, "CSV": { "name": "CSV", "kind": "attribute", "path": "omniread.core.ContentType.CSV", - "signature": "", + "signature": null, "docstring": "Comma-separated-value document content." }, "JSON": { "name": "JSON", "kind": "attribute", "path": "omniread.core.ContentType.JSON", - "signature": "", + "signature": null, "docstring": "JSON document content." }, "XML": { "name": "XML", "kind": "attribute", "path": "omniread.core.ContentType.XML", - "signature": "", + "signature": null, "docstring": "XML document content." } } @@ -96,35 +96,35 @@ "name": "BaseParser", "kind": "class", "path": "omniread.core.BaseParser", - "signature": "", + "signature": "BaseParser(content: Content)", "docstring": "Base interface for all parsers.\n\nNotes:\n **Guarantees:**\n\n - A parser is a self-contained object that owns the `Content` it is\n responsible for interpreting.\n - Consumers may rely on early validation of content compatibility\n and type-stable return values from `parse()`.\n\n **Responsibilities:**\n\n - Implementations must declare supported content types via `supported_types`.\n - Implementations must raise parsing-specific exceptions from `parse()`.\n - Implementations must remain deterministic for a given input.", "members": { "supported_types": { "name": "supported_types", "kind": "attribute", "path": "omniread.core.BaseParser.supported_types", - "signature": "", + "signature": null, "docstring": "Set of content types supported by this parser. An empty set indicates that the parser is content-type agnostic." }, "content": { "name": "content", "kind": "attribute", "path": "omniread.core.BaseParser.content", - "signature": "", + "signature": null, "docstring": null }, "parse": { "name": "parse", "kind": "function", "path": "omniread.core.BaseParser.parse", - "signature": "", + "signature": "parse()", "docstring": "Parse the owned content into structured output.\n\nReturns:\n T:\n Parsed, structured representation.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must fully consume the provided content and\n return a deterministic, structured output." }, "supports": { "name": "supports", "kind": "function", "path": "omniread.core.BaseParser.supports", - "signature": "", + "signature": "supports()", "docstring": "Check whether this parser supports the content's type.\n\nReturns:\n bool:\n True if the content type is supported; False otherwise." } } @@ -133,14 +133,14 @@ "name": "BaseScraper", "kind": "class", "path": "omniread.core.BaseScraper", - "signature": "", + "signature": null, "docstring": "Base interface for all scrapers.\n\nNotes:\n **Responsibilities:**\n\n - A scraper is responsible ONLY for fetching raw content (bytes)\n from a source. It must not interpret or parse it.\n - A scraper is a stateless acquisition component that retrieves raw\n content from a source and returns it as a `Content` object.\n - Scrapers define how content is obtained, not what the content means.\n - Implementations may vary in transport mechanism, authentication\n strategy, retry and backoff behavior.\n\n **Constraints:**\n\n - Implementations must not parse content, modify content semantics,\n or couple scraping logic to a specific parser.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.core.BaseScraper.fetch", - "signature": "", + "signature": "fetch(source: str, *, metadata: Mapping[str, Any] | None = None)", "docstring": "Fetch raw content from the given source.\n\nArgs:\n source (str):\n Location identifier (URL, file path, S3 URI, etc.).\n\n metadata (Mapping[str, Any] | None, optional):\n Optional hints for the scraper (headers, auth, etc.).\n\nReturns:\n Content:\n Content object containing raw bytes and metadata.\n\nRaises:\n Exception:\n Retrieval-specific errors as defined by the implementation.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must retrieve the content referenced by `source`\n and return it as raw bytes wrapped in a `Content` object." } } @@ -152,39 +152,11 @@ "signature": null, "docstring": "# Summary\n\nCanonical content models for OmniRead.\n\nThis module defines the **format-agnostic content representation** used across\nall parsers and scrapers in OmniRead.\n\nThe models defined here represent *what* was extracted, not *how* it was\nretrieved or parsed. Format-specific behavior and metadata must not alter\nthe semantic meaning of these models.", "members": { - "Mapping": { - "name": "Mapping", - "kind": "alias", - "path": "omniread.core.content.Mapping", - "signature": "", - "docstring": null - }, - "dataclass": { - "name": "dataclass", - "kind": "alias", - "path": "omniread.core.content.dataclass", - "signature": "", - "docstring": null - }, - "Enum": { - "name": "Enum", - "kind": "alias", - "path": "omniread.core.content.Enum", - "signature": "", - "docstring": null - }, - "Any": { - "name": "Any", - "kind": "alias", - "path": "omniread.core.content.Any", - "signature": "", - "docstring": null - }, "ContentType": { "name": "ContentType", "kind": "class", "path": "omniread.core.content.ContentType", - "signature": "", + "signature": null, "docstring": "Supported MIME types for extracted content.\n\nNotes:\n **Guarantees:**\n\n - This enum represents the declared or inferred media type of the\n content source.\n - It is primarily used for routing content to the appropriate\n parser or downstream consumer.", "members": { "HTML": { @@ -235,7 +207,7 @@ "name": "Content", "kind": "class", "path": "omniread.core.content.Content", - "signature": "", + "signature": "Content(raw: bytes, source: str, content_type: ContentType | None = ..., metadata: Mapping[str, Any] | None = ...)", "docstring": "Normalized representation of extracted content.\n\nNotes:\n **Responsibilities:**\n\n - A `Content` instance represents a raw content payload along with\n minimal contextual metadata describing its origin and type.\n - This class is the primary exchange format between scrapers,\n parsers, and downstream consumers.", "members": { "raw": { @@ -277,67 +249,39 @@ "signature": null, "docstring": "# Summary\n\nAbstract parsing contracts for OmniRead.\n\nThis module defines the **format-agnostic parser interface** used to transform\nraw content into structured, typed representations.\n\nParsers are responsible for:\n\n- Interpreting a single `Content` instance\n- Validating compatibility with the content type\n- Producing a structured output suitable for downstream consumers\n\nParsers are not responsible for:\n\n- Fetching or acquiring content\n- Performing retries or error recovery\n- Managing multiple content sources", "members": { - "ABC": { - "name": "ABC", - "kind": "alias", - "path": "omniread.core.parser.ABC", - "signature": "", - "docstring": null - }, - "abstractmethod": { - "name": "abstractmethod", - "kind": "alias", - "path": "omniread.core.parser.abstractmethod", - "signature": "", - "docstring": null - }, - "Generic": { - "name": "Generic", - "kind": "alias", - "path": "omniread.core.parser.Generic", - "signature": "", - "docstring": null - }, - "TypeVar": { - "name": "TypeVar", - "kind": "alias", - "path": "omniread.core.parser.TypeVar", - "signature": "", - "docstring": null - }, "Content": { "name": "Content", "kind": "class", "path": "omniread.core.parser.Content", - "signature": "", + "signature": "Content(raw: bytes, source: str, content_type: ContentType | None = ..., metadata: Mapping[str, Any] | None = ...)", "docstring": "Normalized representation of extracted content.\n\nNotes:\n **Responsibilities:**\n\n - A `Content` instance represents a raw content payload along with\n minimal contextual metadata describing its origin and type.\n - This class is the primary exchange format between scrapers,\n parsers, and downstream consumers.", "members": { "raw": { "name": "raw", "kind": "attribute", "path": "omniread.core.parser.Content.raw", - "signature": "", + "signature": null, "docstring": "Raw content bytes as retrieved from the source." }, "source": { "name": "source", "kind": "attribute", "path": "omniread.core.parser.Content.source", - "signature": "", + "signature": null, "docstring": "Identifier of the content origin (URL, file path, or logical name)." }, "content_type": { "name": "content_type", "kind": "attribute", "path": "omniread.core.parser.Content.content_type", - "signature": "", + "signature": null, "docstring": "Optional MIME type of the content, if known." }, "metadata": { "name": "metadata", "kind": "attribute", "path": "omniread.core.parser.Content.metadata", - "signature": "", + "signature": null, "docstring": "Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes)." } } @@ -346,49 +290,49 @@ "name": "ContentType", "kind": "class", "path": "omniread.core.parser.ContentType", - "signature": "", + "signature": null, "docstring": "Supported MIME types for extracted content.\n\nNotes:\n **Guarantees:**\n\n - This enum represents the declared or inferred media type of the\n content source.\n - It is primarily used for routing content to the appropriate\n parser or downstream consumer.", "members": { "HTML": { "name": "HTML", "kind": "attribute", "path": "omniread.core.parser.ContentType.HTML", - "signature": "", + "signature": null, "docstring": "HTML document content." }, "PDF": { "name": "PDF", "kind": "attribute", "path": "omniread.core.parser.ContentType.PDF", - "signature": "", + "signature": null, "docstring": "PDF document content." }, "XLSX": { "name": "XLSX", "kind": "attribute", "path": "omniread.core.parser.ContentType.XLSX", - "signature": "", + "signature": null, "docstring": "Office Open XML spreadsheet (xlsx/xlsm) content." }, "CSV": { "name": "CSV", "kind": "attribute", "path": "omniread.core.parser.ContentType.CSV", - "signature": "", + "signature": null, "docstring": "Comma-separated-value document content." }, "JSON": { "name": "JSON", "kind": "attribute", "path": "omniread.core.parser.ContentType.JSON", - "signature": "", + "signature": null, "docstring": "JSON document content." }, "XML": { "name": "XML", "kind": "attribute", "path": "omniread.core.parser.ContentType.XML", - "signature": "", + "signature": null, "docstring": "XML document content." } } @@ -404,7 +348,7 @@ "name": "BaseParser", "kind": "class", "path": "omniread.core.parser.BaseParser", - "signature": "", + "signature": "BaseParser(content: Content)", "docstring": "Base interface for all parsers.\n\nNotes:\n **Guarantees:**\n\n - A parser is a self-contained object that owns the `Content` it is\n responsible for interpreting.\n - Consumers may rely on early validation of content compatibility\n and type-stable return values from `parse()`.\n\n **Responsibilities:**\n\n - Implementations must declare supported content types via `supported_types`.\n - Implementations must raise parsing-specific exceptions from `parse()`.\n - Implementations must remain deterministic for a given input.", "members": { "supported_types": { @@ -425,14 +369,14 @@ "name": "parse", "kind": "function", "path": "omniread.core.parser.BaseParser.parse", - "signature": "", + "signature": "parse() -> T", "docstring": "Parse the owned content into structured output.\n\nReturns:\n T:\n Parsed, structured representation.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must fully consume the provided content and\n return a deterministic, structured output." }, "supports": { "name": "supports", "kind": "function", "path": "omniread.core.parser.BaseParser.supports", - "signature": "", + "signature": "supports() -> bool", "docstring": "Check whether this parser supports the content's type.\n\nReturns:\n bool:\n True if the content type is supported; False otherwise." } } @@ -446,67 +390,39 @@ "signature": null, "docstring": "# Summary\n\nAbstract scraping contracts for OmniRead.\n\nThis module defines the **format-agnostic scraper interface** responsible for\nacquiring raw content from external sources.\n\nScrapers are responsible for:\n\n- Locating and retrieving raw content bytes\n- Attaching minimal contextual metadata\n- Returning normalized `Content` objects\n\nScrapers are explicitly NOT responsible for:\n\n- Parsing or interpreting content\n- Inferring structure or semantics\n- Performing content-type specific processing\n\nAll interpretation must be delegated to parsers.", "members": { - "ABC": { - "name": "ABC", - "kind": "alias", - "path": "omniread.core.scraper.ABC", - "signature": "", - "docstring": null - }, - "abstractmethod": { - "name": "abstractmethod", - "kind": "alias", - "path": "omniread.core.scraper.abstractmethod", - "signature": "", - "docstring": null - }, - "Mapping": { - "name": "Mapping", - "kind": "alias", - "path": "omniread.core.scraper.Mapping", - "signature": "", - "docstring": null - }, - "Any": { - "name": "Any", - "kind": "alias", - "path": "omniread.core.scraper.Any", - "signature": "", - "docstring": null - }, "Content": { "name": "Content", "kind": "class", "path": "omniread.core.scraper.Content", - "signature": "", + "signature": "Content(raw: bytes, source: str, content_type: ContentType | None = ..., metadata: Mapping[str, Any] | None = ...)", "docstring": "Normalized representation of extracted content.\n\nNotes:\n **Responsibilities:**\n\n - A `Content` instance represents a raw content payload along with\n minimal contextual metadata describing its origin and type.\n - This class is the primary exchange format between scrapers,\n parsers, and downstream consumers.", "members": { "raw": { "name": "raw", "kind": "attribute", "path": "omniread.core.scraper.Content.raw", - "signature": "", + "signature": null, "docstring": "Raw content bytes as retrieved from the source." }, "source": { "name": "source", "kind": "attribute", "path": "omniread.core.scraper.Content.source", - "signature": "", + "signature": null, "docstring": "Identifier of the content origin (URL, file path, or logical name)." }, "content_type": { "name": "content_type", "kind": "attribute", "path": "omniread.core.scraper.Content.content_type", - "signature": "", + "signature": null, "docstring": "Optional MIME type of the content, if known." }, "metadata": { "name": "metadata", "kind": "attribute", "path": "omniread.core.scraper.Content.metadata", - "signature": "", + "signature": null, "docstring": "Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes)." } } @@ -515,14 +431,14 @@ "name": "BaseScraper", "kind": "class", "path": "omniread.core.scraper.BaseScraper", - "signature": "", + "signature": null, "docstring": "Base interface for all scrapers.\n\nNotes:\n **Responsibilities:**\n\n - A scraper is responsible ONLY for fetching raw content (bytes)\n from a source. It must not interpret or parse it.\n - A scraper is a stateless acquisition component that retrieves raw\n content from a source and returns it as a `Content` object.\n - Scrapers define how content is obtained, not what the content means.\n - Implementations may vary in transport mechanism, authentication\n strategy, retry and backoff behavior.\n\n **Constraints:**\n\n - Implementations must not parse content, modify content semantics,\n or couple scraping logic to a specific parser.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.core.scraper.BaseScraper.fetch", - "signature": "", + "signature": "fetch(source: str, *, metadata: Mapping[str, Any] | None = None) -> Content", "docstring": "Fetch raw content from the given source.\n\nArgs:\n source (str):\n Location identifier (URL, file path, S3 URI, etc.).\n\n metadata (Mapping[str, Any] | None, optional):\n Optional hints for the scraper (headers, auth, etc.).\n\nReturns:\n Content:\n Content object containing raw bytes and metadata.\n\nRaises:\n Exception:\n Retrieval-specific errors as defined by the implementation.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must retrieve the content referenced by `source`\n and return it as raw bytes wrapped in a `Content` object." } } diff --git a/mcp/omniread/modules/omniread.core.parser.json b/mcp/omniread/modules/omniread.core.parser.json index c10e259..88c67ce 100644 --- a/mcp/omniread/modules/omniread.core.parser.json +++ b/mcp/omniread/modules/omniread.core.parser.json @@ -4,67 +4,39 @@ "path": "omniread.core.parser", "docstring": "# Summary\n\nAbstract parsing contracts for OmniRead.\n\nThis module defines the **format-agnostic parser interface** used to transform\nraw content into structured, typed representations.\n\nParsers are responsible for:\n\n- Interpreting a single `Content` instance\n- Validating compatibility with the content type\n- Producing a structured output suitable for downstream consumers\n\nParsers are not responsible for:\n\n- Fetching or acquiring content\n- Performing retries or error recovery\n- Managing multiple content sources", "objects": { - "ABC": { - "name": "ABC", - "kind": "alias", - "path": "omniread.core.parser.ABC", - "signature": "", - "docstring": null - }, - "abstractmethod": { - "name": "abstractmethod", - "kind": "alias", - "path": "omniread.core.parser.abstractmethod", - "signature": "", - "docstring": null - }, - "Generic": { - "name": "Generic", - "kind": "alias", - "path": "omniread.core.parser.Generic", - "signature": "", - "docstring": null - }, - "TypeVar": { - "name": "TypeVar", - "kind": "alias", - "path": "omniread.core.parser.TypeVar", - "signature": "", - "docstring": null - }, "Content": { "name": "Content", "kind": "class", "path": "omniread.core.parser.Content", - "signature": "", + "signature": "Content(raw: bytes, source: str, content_type: ContentType | None = ..., metadata: Mapping[str, Any] | None = ...)", "docstring": "Normalized representation of extracted content.\n\nNotes:\n **Responsibilities:**\n\n - A `Content` instance represents a raw content payload along with\n minimal contextual metadata describing its origin and type.\n - This class is the primary exchange format between scrapers,\n parsers, and downstream consumers.", "members": { "raw": { "name": "raw", "kind": "attribute", "path": "omniread.core.parser.Content.raw", - "signature": "", + "signature": null, "docstring": "Raw content bytes as retrieved from the source." }, "source": { "name": "source", "kind": "attribute", "path": "omniread.core.parser.Content.source", - "signature": "", + "signature": null, "docstring": "Identifier of the content origin (URL, file path, or logical name)." }, "content_type": { "name": "content_type", "kind": "attribute", "path": "omniread.core.parser.Content.content_type", - "signature": "", + "signature": null, "docstring": "Optional MIME type of the content, if known." }, "metadata": { "name": "metadata", "kind": "attribute", "path": "omniread.core.parser.Content.metadata", - "signature": "", + "signature": null, "docstring": "Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes)." } } @@ -73,49 +45,49 @@ "name": "ContentType", "kind": "class", "path": "omniread.core.parser.ContentType", - "signature": "", + "signature": null, "docstring": "Supported MIME types for extracted content.\n\nNotes:\n **Guarantees:**\n\n - This enum represents the declared or inferred media type of the\n content source.\n - It is primarily used for routing content to the appropriate\n parser or downstream consumer.", "members": { "HTML": { "name": "HTML", "kind": "attribute", "path": "omniread.core.parser.ContentType.HTML", - "signature": "", + "signature": null, "docstring": "HTML document content." }, "PDF": { "name": "PDF", "kind": "attribute", "path": "omniread.core.parser.ContentType.PDF", - "signature": "", + "signature": null, "docstring": "PDF document content." }, "XLSX": { "name": "XLSX", "kind": "attribute", "path": "omniread.core.parser.ContentType.XLSX", - "signature": "", + "signature": null, "docstring": "Office Open XML spreadsheet (xlsx/xlsm) content." }, "CSV": { "name": "CSV", "kind": "attribute", "path": "omniread.core.parser.ContentType.CSV", - "signature": "", + "signature": null, "docstring": "Comma-separated-value document content." }, "JSON": { "name": "JSON", "kind": "attribute", "path": "omniread.core.parser.ContentType.JSON", - "signature": "", + "signature": null, "docstring": "JSON document content." }, "XML": { "name": "XML", "kind": "attribute", "path": "omniread.core.parser.ContentType.XML", - "signature": "", + "signature": null, "docstring": "XML document content." } } @@ -131,7 +103,7 @@ "name": "BaseParser", "kind": "class", "path": "omniread.core.parser.BaseParser", - "signature": "", + "signature": "BaseParser(content: Content)", "docstring": "Base interface for all parsers.\n\nNotes:\n **Guarantees:**\n\n - A parser is a self-contained object that owns the `Content` it is\n responsible for interpreting.\n - Consumers may rely on early validation of content compatibility\n and type-stable return values from `parse()`.\n\n **Responsibilities:**\n\n - Implementations must declare supported content types via `supported_types`.\n - Implementations must raise parsing-specific exceptions from `parse()`.\n - Implementations must remain deterministic for a given input.", "members": { "supported_types": { @@ -152,14 +124,14 @@ "name": "parse", "kind": "function", "path": "omniread.core.parser.BaseParser.parse", - "signature": "", + "signature": "parse() -> T", "docstring": "Parse the owned content into structured output.\n\nReturns:\n T:\n Parsed, structured representation.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must fully consume the provided content and\n return a deterministic, structured output." }, "supports": { "name": "supports", "kind": "function", "path": "omniread.core.parser.BaseParser.supports", - "signature": "", + "signature": "supports() -> bool", "docstring": "Check whether this parser supports the content's type.\n\nReturns:\n bool:\n True if the content type is supported; False otherwise." } } diff --git a/mcp/omniread/modules/omniread.core.scraper.json b/mcp/omniread/modules/omniread.core.scraper.json index 7945331..5bb6bde 100644 --- a/mcp/omniread/modules/omniread.core.scraper.json +++ b/mcp/omniread/modules/omniread.core.scraper.json @@ -4,67 +4,39 @@ "path": "omniread.core.scraper", "docstring": "# Summary\n\nAbstract scraping contracts for OmniRead.\n\nThis module defines the **format-agnostic scraper interface** responsible for\nacquiring raw content from external sources.\n\nScrapers are responsible for:\n\n- Locating and retrieving raw content bytes\n- Attaching minimal contextual metadata\n- Returning normalized `Content` objects\n\nScrapers are explicitly NOT responsible for:\n\n- Parsing or interpreting content\n- Inferring structure or semantics\n- Performing content-type specific processing\n\nAll interpretation must be delegated to parsers.", "objects": { - "ABC": { - "name": "ABC", - "kind": "alias", - "path": "omniread.core.scraper.ABC", - "signature": "", - "docstring": null - }, - "abstractmethod": { - "name": "abstractmethod", - "kind": "alias", - "path": "omniread.core.scraper.abstractmethod", - "signature": "", - "docstring": null - }, - "Mapping": { - "name": "Mapping", - "kind": "alias", - "path": "omniread.core.scraper.Mapping", - "signature": "", - "docstring": null - }, - "Any": { - "name": "Any", - "kind": "alias", - "path": "omniread.core.scraper.Any", - "signature": "", - "docstring": null - }, "Content": { "name": "Content", "kind": "class", "path": "omniread.core.scraper.Content", - "signature": "", + "signature": "Content(raw: bytes, source: str, content_type: ContentType | None = ..., metadata: Mapping[str, Any] | None = ...)", "docstring": "Normalized representation of extracted content.\n\nNotes:\n **Responsibilities:**\n\n - A `Content` instance represents a raw content payload along with\n minimal contextual metadata describing its origin and type.\n - This class is the primary exchange format between scrapers,\n parsers, and downstream consumers.", "members": { "raw": { "name": "raw", "kind": "attribute", "path": "omniread.core.scraper.Content.raw", - "signature": "", + "signature": null, "docstring": "Raw content bytes as retrieved from the source." }, "source": { "name": "source", "kind": "attribute", "path": "omniread.core.scraper.Content.source", - "signature": "", + "signature": null, "docstring": "Identifier of the content origin (URL, file path, or logical name)." }, "content_type": { "name": "content_type", "kind": "attribute", "path": "omniread.core.scraper.Content.content_type", - "signature": "", + "signature": null, "docstring": "Optional MIME type of the content, if known." }, "metadata": { "name": "metadata", "kind": "attribute", "path": "omniread.core.scraper.Content.metadata", - "signature": "", + "signature": null, "docstring": "Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes)." } } @@ -73,14 +45,14 @@ "name": "BaseScraper", "kind": "class", "path": "omniread.core.scraper.BaseScraper", - "signature": "", + "signature": null, "docstring": "Base interface for all scrapers.\n\nNotes:\n **Responsibilities:**\n\n - A scraper is responsible ONLY for fetching raw content (bytes)\n from a source. It must not interpret or parse it.\n - A scraper is a stateless acquisition component that retrieves raw\n content from a source and returns it as a `Content` object.\n - Scrapers define how content is obtained, not what the content means.\n - Implementations may vary in transport mechanism, authentication\n strategy, retry and backoff behavior.\n\n **Constraints:**\n\n - Implementations must not parse content, modify content semantics,\n or couple scraping logic to a specific parser.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.core.scraper.BaseScraper.fetch", - "signature": "", + "signature": "fetch(source: str, *, metadata: Mapping[str, Any] | None = None) -> Content", "docstring": "Fetch raw content from the given source.\n\nArgs:\n source (str):\n Location identifier (URL, file path, S3 URI, etc.).\n\n metadata (Mapping[str, Any] | None, optional):\n Optional hints for the scraper (headers, auth, etc.).\n\nReturns:\n Content:\n Content object containing raw bytes and metadata.\n\nRaises:\n Exception:\n Retrieval-specific errors as defined by the implementation.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must retrieve the content referenced by `source`\n and return it as raw bytes wrapped in a `Content` object." } } diff --git a/mcp/omniread/modules/omniread.csv.client.json b/mcp/omniread/modules/omniread.csv.client.json index af0c03e..7d36d79 100644 --- a/mcp/omniread/modules/omniread.csv.client.json +++ b/mcp/omniread/modules/omniread.csv.client.json @@ -4,46 +4,18 @@ "path": "omniread.csv.client", "docstring": "# Summary\n\nCSV client abstractions for OmniRead.\n\nThis module defines the **client layer** responsible for retrieving raw\ncomma-separated-value document bytes from a concrete backing store.\n\nClients provide low-level access to csv binaries and are intentionally\ndecoupled from scraping and parsing logic. They do not perform validation,\ninterpretation, or content extraction.\n\nTypical backing stores include:\n\n- Local filesystems\n- Object storage (S3, GCS, etc.)\n- Network file systems", "objects": { - "ABC": { - "name": "ABC", - "kind": "alias", - "path": "omniread.csv.client.ABC", - "signature": "", - "docstring": null - }, - "abstractmethod": { - "name": "abstractmethod", - "kind": "alias", - "path": "omniread.csv.client.abstractmethod", - "signature": "", - "docstring": null - }, - "Path": { - "name": "Path", - "kind": "alias", - "path": "omniread.csv.client.Path", - "signature": "", - "docstring": null - }, - "Any": { - "name": "Any", - "kind": "alias", - "path": "omniread.csv.client.Any", - "signature": "", - "docstring": null - }, "BaseCsvClient": { "name": "BaseCsvClient", "kind": "class", "path": "omniread.csv.client.BaseCsvClient", - "signature": "", + "signature": null, "docstring": "Abstract client responsible for retrieving csv bytes.\n\nRetrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).\n\nNotes:\n **Responsibilities:**\n\n - Implementations must accept a source identifier appropriate to\n the backing store.\n - Return the full csv binary payload.\n - Raise retrieval-specific errors on failure.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.csv.client.BaseCsvClient.fetch", - "signature": "", + "signature": "fetch(source: Any) -> bytes", "docstring": "Fetch raw csv bytes from the given source.\n\nArgs:\n source (Any):\n Identifier of the csv location, such as a file path,\n object storage key, or remote reference.\n\nReturns:\n bytes:\n Raw csv bytes.\n\nRaises:\n Exception:\n Retrieval-specific errors defined by the implementation." } } @@ -52,14 +24,14 @@ "name": "FileSystemCsvClient", "kind": "class", "path": "omniread.csv.client.FileSystemCsvClient", - "signature": "", + "signature": null, "docstring": "CSV client that reads from the local filesystem.\n\nNotes:\n **Guarantees:**\n\n - This client reads csv files directly from the disk and\n returns their raw binary contents.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.csv.client.FileSystemCsvClient.fetch", - "signature": "", + "signature": "fetch(path: Path) -> bytes", "docstring": "Read a csv file from the local filesystem.\n\nArgs:\n path (Path):\n Filesystem path to the csv file.\n\nReturns:\n bytes:\n Raw csv bytes.\n\nRaises:\n FileNotFoundError:\n If the path does not exist.\n ValueError:\n If the path exists but is not a file." } } diff --git a/mcp/omniread/modules/omniread.csv.json b/mcp/omniread/modules/omniread.csv.json index 46215e9..0dfa9aa 100644 --- a/mcp/omniread/modules/omniread.csv.json +++ b/mcp/omniread/modules/omniread.csv.json @@ -8,14 +8,14 @@ "name": "BaseCsvClient", "kind": "class", "path": "omniread.csv.BaseCsvClient", - "signature": "", + "signature": null, "docstring": "Abstract client responsible for retrieving csv bytes.\n\nRetrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).\n\nNotes:\n **Responsibilities:**\n\n - Implementations must accept a source identifier appropriate to\n the backing store.\n - Return the full csv binary payload.\n - Raise retrieval-specific errors on failure.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.csv.BaseCsvClient.fetch", - "signature": "", + "signature": "fetch(source: Any)", "docstring": "Fetch raw csv bytes from the given source.\n\nArgs:\n source (Any):\n Identifier of the csv location, such as a file path,\n object storage key, or remote reference.\n\nReturns:\n bytes:\n Raw csv bytes.\n\nRaises:\n Exception:\n Retrieval-specific errors defined by the implementation." } } @@ -24,14 +24,14 @@ "name": "FileSystemCsvClient", "kind": "class", "path": "omniread.csv.FileSystemCsvClient", - "signature": "", + "signature": null, "docstring": "CSV client that reads from the local filesystem.\n\nNotes:\n **Guarantees:**\n\n - This client reads csv files directly from the disk and\n returns their raw binary contents.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.csv.FileSystemCsvClient.fetch", - "signature": "", + "signature": "fetch(path: Path)", "docstring": "Read a csv file from the local filesystem.\n\nArgs:\n path (Path):\n Filesystem path to the csv file.\n\nReturns:\n bytes:\n Raw csv bytes.\n\nRaises:\n FileNotFoundError:\n If the path does not exist.\n ValueError:\n If the path exists but is not a file." } } @@ -40,21 +40,21 @@ "name": "CsvParser", "kind": "class", "path": "omniread.csv.CsvParser", - "signature": "", + "signature": "CsvParser(content: Content)", "docstring": "Generic csv parser producing string rows from the document.\n\nNotes:\n **Responsibilities:**\n\n - Decode the payload (UTF-8 with BOM support, Latin-1 fallback).\n - Detect the delimiter from a leading sample (`,` `;` tab `|`),\n defaulting to `,`.\n - Normalize cells into deterministic stripped string values.\n - Expose row extraction helpers mirroring `XlsxParser.rows`.\n\n **Constraints:**\n\n - All values are strings; consumers requiring typed values must\n convert on their side.\n - Quoted fields containing delimiters/newlines are handled by\n the standard ``csv`` module.", "members": { "parse": { "name": "parse", "kind": "function", "path": "omniread.csv.CsvParser.parse", - "signature": "", + "signature": "parse()", "docstring": "Parse the document into normalized string rows.\n\nReturns:\n list[list[str]]:\n Rows of the document." }, "rows": { "name": "rows", "kind": "function", "path": "omniread.csv.CsvParser.rows", - "signature": "", + "signature": "rows(*, skip_empty: bool = True)", "docstring": "Extract normalized string rows from the document.\n\nArgs:\n skip_empty (bool):\n When True (default), rows whose cells are all blank are\n omitted.\n\nReturns:\n list[list[str]]:\n Normalized rows; trailing blank cells are trimmed per row." } } @@ -63,21 +63,21 @@ "name": "CsvParserBase", "kind": "class", "path": "omniread.csv.CsvParserBase", - "signature": "", + "signature": null, "docstring": "Base csv parser.\n\nNotes:\n **Responsibilities:**\n\n - This class enforces csv content-type compatibility and provides\n the extension point for implementing concrete csv parsing\n strategies.\n\n **Constraints:**\n\n - Concrete implementations must define the output type `T` and\n implement the `parse()` method.", "members": { "supported_types": { "name": "supported_types", "kind": "attribute", "path": "omniread.csv.CsvParserBase.supported_types", - "signature": "", + "signature": null, "docstring": "Set of content types supported by this parser (CSV only)." }, "parse": { "name": "parse", "kind": "function", "path": "omniread.csv.CsvParserBase.parse", - "signature": "", + "signature": "parse()", "docstring": "Parse csv content into a structured output.\n\nReturns:\n T:\n Parsed representation of type `T`.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation." } } @@ -86,14 +86,14 @@ "name": "CsvScraper", "kind": "class", "path": "omniread.csv.CsvScraper", - "signature": "", + "signature": "CsvScraper(*, client: BaseCsvClient)", "docstring": "Scraper for csv documents.\n\nNotes:\n **Responsibilities:**\n\n - Fetch raw csv bytes via the configured client.\n - Wrap the payload in a canonical `Content` instance with the\n CSV content type and source identifier.\n\n **Constraints:**\n\n - The scraper does not perform parsing or interpretation.\n - Does not assume a specific storage backend.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.csv.CsvScraper.fetch", - "signature": "", + "signature": "fetch(source: Any, *, metadata: Mapping[str, Any] | None = None)", "docstring": "Fetch a csv document from the given source.\n\nArgs:\n source (Any):\n Identifier of the csv source as understood by the\n configured client.\n metadata (Mapping[str, Any] | None, optional):\n Optional metadata to attach to the returned content.\n\nReturns:\n Content:\n A `Content` instance containing raw csv bytes, source\n identifier, CSV content type, and optional metadata.\n\nRaises:\n Exception:\n Retrieval-specific errors raised by the client." } } @@ -105,46 +105,18 @@ "signature": null, "docstring": "# Summary\n\nCSV client abstractions for OmniRead.\n\nThis module defines the **client layer** responsible for retrieving raw\ncomma-separated-value document bytes from a concrete backing store.\n\nClients provide low-level access to csv binaries and are intentionally\ndecoupled from scraping and parsing logic. They do not perform validation,\ninterpretation, or content extraction.\n\nTypical backing stores include:\n\n- Local filesystems\n- Object storage (S3, GCS, etc.)\n- Network file systems", "members": { - "ABC": { - "name": "ABC", - "kind": "alias", - "path": "omniread.csv.client.ABC", - "signature": "", - "docstring": null - }, - "abstractmethod": { - "name": "abstractmethod", - "kind": "alias", - "path": "omniread.csv.client.abstractmethod", - "signature": "", - "docstring": null - }, - "Path": { - "name": "Path", - "kind": "alias", - "path": "omniread.csv.client.Path", - "signature": "", - "docstring": null - }, - "Any": { - "name": "Any", - "kind": "alias", - "path": "omniread.csv.client.Any", - "signature": "", - "docstring": null - }, "BaseCsvClient": { "name": "BaseCsvClient", "kind": "class", "path": "omniread.csv.client.BaseCsvClient", - "signature": "", + "signature": null, "docstring": "Abstract client responsible for retrieving csv bytes.\n\nRetrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).\n\nNotes:\n **Responsibilities:**\n\n - Implementations must accept a source identifier appropriate to\n the backing store.\n - Return the full csv binary payload.\n - Raise retrieval-specific errors on failure.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.csv.client.BaseCsvClient.fetch", - "signature": "", + "signature": "fetch(source: Any) -> bytes", "docstring": "Fetch raw csv bytes from the given source.\n\nArgs:\n source (Any):\n Identifier of the csv location, such as a file path,\n object storage key, or remote reference.\n\nReturns:\n bytes:\n Raw csv bytes.\n\nRaises:\n Exception:\n Retrieval-specific errors defined by the implementation." } } @@ -153,14 +125,14 @@ "name": "FileSystemCsvClient", "kind": "class", "path": "omniread.csv.client.FileSystemCsvClient", - "signature": "", + "signature": null, "docstring": "CSV client that reads from the local filesystem.\n\nNotes:\n **Guarantees:**\n\n - This client reads csv files directly from the disk and\n returns their raw binary contents.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.csv.client.FileSystemCsvClient.fetch", - "signature": "", + "signature": "fetch(path: Path) -> bytes", "docstring": "Read a csv file from the local filesystem.\n\nArgs:\n path (Path):\n Filesystem path to the csv file.\n\nReturns:\n bytes:\n Raw csv bytes.\n\nRaises:\n FileNotFoundError:\n If the path does not exist.\n ValueError:\n If the path exists but is not a file." } } @@ -174,60 +146,39 @@ "signature": null, "docstring": "# Summary\n\nCSV parser implementations for OmniRead.\n\nThis module provides a concrete, generic parser for comma-separated-value\ndocuments. It exposes records as lists of string cells so downstream\nconsumers can interpret tabular content without depending on the ``csv``\nmodule directly.\n\nThe parser is intentionally statement-agnostic: it performs no header\ndetection or column interpretation beyond basic cell normalization and\ndelimiter detection.", "members": { - "Sniffer": { - "name": "Sniffer", - "kind": "alias", - "path": "omniread.csv.parser.Sniffer", - "signature": "", - "docstring": null - }, - "reader": { - "name": "reader", - "kind": "alias", - "path": "omniread.csv.parser.reader", - "signature": "", - "docstring": null - }, - "StringIO": { - "name": "StringIO", - "kind": "alias", - "path": "omniread.csv.parser.StringIO", - "signature": "", - "docstring": null - }, "Content": { "name": "Content", "kind": "class", "path": "omniread.csv.parser.Content", - "signature": "", + "signature": "Content(raw: bytes, source: str, content_type: ContentType | None = ..., metadata: Mapping[str, Any] | None = ...)", "docstring": "Normalized representation of extracted content.\n\nNotes:\n **Responsibilities:**\n\n - A `Content` instance represents a raw content payload along with\n minimal contextual metadata describing its origin and type.\n - This class is the primary exchange format between scrapers,\n parsers, and downstream consumers.", "members": { "raw": { "name": "raw", "kind": "attribute", "path": "omniread.csv.parser.Content.raw", - "signature": "", + "signature": null, "docstring": "Raw content bytes as retrieved from the source." }, "source": { "name": "source", "kind": "attribute", "path": "omniread.csv.parser.Content.source", - "signature": "", + "signature": null, "docstring": "Identifier of the content origin (URL, file path, or logical name)." }, "content_type": { "name": "content_type", "kind": "attribute", "path": "omniread.csv.parser.Content.content_type", - "signature": "", + "signature": null, "docstring": "Optional MIME type of the content, if known." }, "metadata": { "name": "metadata", "kind": "attribute", "path": "omniread.csv.parser.Content.metadata", - "signature": "", + "signature": null, "docstring": "Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes)." } } @@ -236,21 +187,21 @@ "name": "CsvParserBase", "kind": "class", "path": "omniread.csv.parser.CsvParserBase", - "signature": "", + "signature": null, "docstring": "Base csv parser.\n\nNotes:\n **Responsibilities:**\n\n - This class enforces csv content-type compatibility and provides\n the extension point for implementing concrete csv parsing\n strategies.\n\n **Constraints:**\n\n - Concrete implementations must define the output type `T` and\n implement the `parse()` method.", "members": { "supported_types": { "name": "supported_types", "kind": "attribute", "path": "omniread.csv.parser.CsvParserBase.supported_types", - "signature": "", + "signature": null, "docstring": "Set of content types supported by this parser (CSV only)." }, "parse": { "name": "parse", "kind": "function", "path": "omniread.csv.parser.CsvParserBase.parse", - "signature": "", + "signature": "parse()", "docstring": "Parse csv content into a structured output.\n\nReturns:\n T:\n Parsed representation of type `T`.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation." } } @@ -259,21 +210,21 @@ "name": "CsvParser", "kind": "class", "path": "omniread.csv.parser.CsvParser", - "signature": "", + "signature": "CsvParser(content: Content)", "docstring": "Generic csv parser producing string rows from the document.\n\nNotes:\n **Responsibilities:**\n\n - Decode the payload (UTF-8 with BOM support, Latin-1 fallback).\n - Detect the delimiter from a leading sample (`,` `;` tab `|`),\n defaulting to `,`.\n - Normalize cells into deterministic stripped string values.\n - Expose row extraction helpers mirroring `XlsxParser.rows`.\n\n **Constraints:**\n\n - All values are strings; consumers requiring typed values must\n convert on their side.\n - Quoted fields containing delimiters/newlines are handled by\n the standard ``csv`` module.", "members": { "parse": { "name": "parse", "kind": "function", "path": "omniread.csv.parser.CsvParser.parse", - "signature": "", + "signature": "parse() -> list[list[str]]", "docstring": "Parse the document into normalized string rows.\n\nReturns:\n list[list[str]]:\n Rows of the document." }, "rows": { "name": "rows", "kind": "function", "path": "omniread.csv.parser.CsvParser.rows", - "signature": "", + "signature": "rows(*, skip_empty: bool = True) -> list[list[str]]", "docstring": "Extract normalized string rows from the document.\n\nArgs:\n skip_empty (bool):\n When True (default), rows whose cells are all blank are\n omitted.\n\nReturns:\n list[list[str]]:\n Normalized rows; trailing blank cells are trimmed per row." } } @@ -287,74 +238,53 @@ "signature": null, "docstring": "# Summary\n\nCSV parser base implementation for OmniRead.\n\nThis module defines the **CSV-specific parser contract**, extending the\nformat-agnostic `BaseParser` with constraints appropriate for\ncomma-separated-value documents.", "members": { - "abstractmethod": { - "name": "abstractmethod", - "kind": "alias", - "path": "omniread.csv.parser_base.abstractmethod", - "signature": "", - "docstring": null - }, - "Generic": { - "name": "Generic", - "kind": "alias", - "path": "omniread.csv.parser_base.Generic", - "signature": "", - "docstring": null - }, - "TypeVar": { - "name": "TypeVar", - "kind": "alias", - "path": "omniread.csv.parser_base.TypeVar", - "signature": "", - "docstring": null - }, "ContentType": { "name": "ContentType", "kind": "class", "path": "omniread.csv.parser_base.ContentType", - "signature": "", + "signature": null, "docstring": "Supported MIME types for extracted content.\n\nNotes:\n **Guarantees:**\n\n - This enum represents the declared or inferred media type of the\n content source.\n - It is primarily used for routing content to the appropriate\n parser or downstream consumer.", "members": { "HTML": { "name": "HTML", "kind": "attribute", "path": "omniread.csv.parser_base.ContentType.HTML", - "signature": "", + "signature": null, "docstring": "HTML document content." }, "PDF": { "name": "PDF", "kind": "attribute", "path": "omniread.csv.parser_base.ContentType.PDF", - "signature": "", + "signature": null, "docstring": "PDF document content." }, "XLSX": { "name": "XLSX", "kind": "attribute", "path": "omniread.csv.parser_base.ContentType.XLSX", - "signature": "", + "signature": null, "docstring": "Office Open XML spreadsheet (xlsx/xlsm) content." }, "CSV": { "name": "CSV", "kind": "attribute", "path": "omniread.csv.parser_base.ContentType.CSV", - "signature": "", + "signature": null, "docstring": "Comma-separated-value document content." }, "JSON": { "name": "JSON", "kind": "attribute", "path": "omniread.csv.parser_base.ContentType.JSON", - "signature": "", + "signature": null, "docstring": "JSON document content." }, "XML": { "name": "XML", "kind": "attribute", "path": "omniread.csv.parser_base.ContentType.XML", - "signature": "", + "signature": null, "docstring": "XML document content." } } @@ -363,35 +293,35 @@ "name": "BaseParser", "kind": "class", "path": "omniread.csv.parser_base.BaseParser", - "signature": "", + "signature": "BaseParser(content: Content)", "docstring": "Base interface for all parsers.\n\nNotes:\n **Guarantees:**\n\n - A parser is a self-contained object that owns the `Content` it is\n responsible for interpreting.\n - Consumers may rely on early validation of content compatibility\n and type-stable return values from `parse()`.\n\n **Responsibilities:**\n\n - Implementations must declare supported content types via `supported_types`.\n - Implementations must raise parsing-specific exceptions from `parse()`.\n - Implementations must remain deterministic for a given input.", "members": { "supported_types": { "name": "supported_types", "kind": "attribute", "path": "omniread.csv.parser_base.BaseParser.supported_types", - "signature": "", + "signature": null, "docstring": "Set of content types supported by this parser. An empty set indicates that the parser is content-type agnostic." }, "content": { "name": "content", "kind": "attribute", "path": "omniread.csv.parser_base.BaseParser.content", - "signature": "", + "signature": null, "docstring": null }, "parse": { "name": "parse", "kind": "function", "path": "omniread.csv.parser_base.BaseParser.parse", - "signature": "", + "signature": "parse()", "docstring": "Parse the owned content into structured output.\n\nReturns:\n T:\n Parsed, structured representation.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must fully consume the provided content and\n return a deterministic, structured output." }, "supports": { "name": "supports", "kind": "function", "path": "omniread.csv.parser_base.BaseParser.supports", - "signature": "", + "signature": "supports()", "docstring": "Check whether this parser supports the content's type.\n\nReturns:\n bool:\n True if the content type is supported; False otherwise." } } @@ -407,7 +337,7 @@ "name": "CsvParserBase", "kind": "class", "path": "omniread.csv.parser_base.CsvParserBase", - "signature": "", + "signature": null, "docstring": "Base csv parser.\n\nNotes:\n **Responsibilities:**\n\n - This class enforces csv content-type compatibility and provides\n the extension point for implementing concrete csv parsing\n strategies.\n\n **Constraints:**\n\n - Concrete implementations must define the output type `T` and\n implement the `parse()` method.", "members": { "supported_types": { @@ -421,7 +351,7 @@ "name": "parse", "kind": "function", "path": "omniread.csv.parser_base.CsvParserBase.parse", - "signature": "", + "signature": "parse() -> T", "docstring": "Parse csv content into a structured output.\n\nReturns:\n T:\n Parsed representation of type `T`.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation." } } @@ -435,53 +365,39 @@ "signature": null, "docstring": "# Summary\n\nCSV scraper for OmniRead.\n\nThis module defines the scraper responsible for acquiring raw\ncomma-separated-value document content from a backing store via a\nconfigured client.\n\nThe scraper does not interpret or parse the acquired bytes; it wraps them in\nthe canonical `Content` model.", "members": { - "Mapping": { - "name": "Mapping", - "kind": "alias", - "path": "omniread.csv.scraper.Mapping", - "signature": "", - "docstring": null - }, - "Any": { - "name": "Any", - "kind": "alias", - "path": "omniread.csv.scraper.Any", - "signature": "", - "docstring": null - }, "Content": { "name": "Content", "kind": "class", "path": "omniread.csv.scraper.Content", - "signature": "", + "signature": "Content(raw: bytes, source: str, content_type: ContentType | None = ..., metadata: Mapping[str, Any] | None = ...)", "docstring": "Normalized representation of extracted content.\n\nNotes:\n **Responsibilities:**\n\n - A `Content` instance represents a raw content payload along with\n minimal contextual metadata describing its origin and type.\n - This class is the primary exchange format between scrapers,\n parsers, and downstream consumers.", "members": { "raw": { "name": "raw", "kind": "attribute", "path": "omniread.csv.scraper.Content.raw", - "signature": "", + "signature": null, "docstring": "Raw content bytes as retrieved from the source." }, "source": { "name": "source", "kind": "attribute", "path": "omniread.csv.scraper.Content.source", - "signature": "", + "signature": null, "docstring": "Identifier of the content origin (URL, file path, or logical name)." }, "content_type": { "name": "content_type", "kind": "attribute", "path": "omniread.csv.scraper.Content.content_type", - "signature": "", + "signature": null, "docstring": "Optional MIME type of the content, if known." }, "metadata": { "name": "metadata", "kind": "attribute", "path": "omniread.csv.scraper.Content.metadata", - "signature": "", + "signature": null, "docstring": "Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes)." } } @@ -490,49 +406,49 @@ "name": "ContentType", "kind": "class", "path": "omniread.csv.scraper.ContentType", - "signature": "", + "signature": null, "docstring": "Supported MIME types for extracted content.\n\nNotes:\n **Guarantees:**\n\n - This enum represents the declared or inferred media type of the\n content source.\n - It is primarily used for routing content to the appropriate\n parser or downstream consumer.", "members": { "HTML": { "name": "HTML", "kind": "attribute", "path": "omniread.csv.scraper.ContentType.HTML", - "signature": "", + "signature": null, "docstring": "HTML document content." }, "PDF": { "name": "PDF", "kind": "attribute", "path": "omniread.csv.scraper.ContentType.PDF", - "signature": "", + "signature": null, "docstring": "PDF document content." }, "XLSX": { "name": "XLSX", "kind": "attribute", "path": "omniread.csv.scraper.ContentType.XLSX", - "signature": "", + "signature": null, "docstring": "Office Open XML spreadsheet (xlsx/xlsm) content." }, "CSV": { "name": "CSV", "kind": "attribute", "path": "omniread.csv.scraper.ContentType.CSV", - "signature": "", + "signature": null, "docstring": "Comma-separated-value document content." }, "JSON": { "name": "JSON", "kind": "attribute", "path": "omniread.csv.scraper.ContentType.JSON", - "signature": "", + "signature": null, "docstring": "JSON document content." }, "XML": { "name": "XML", "kind": "attribute", "path": "omniread.csv.scraper.ContentType.XML", - "signature": "", + "signature": null, "docstring": "XML document content." } } @@ -541,14 +457,14 @@ "name": "BaseCsvClient", "kind": "class", "path": "omniread.csv.scraper.BaseCsvClient", - "signature": "", + "signature": null, "docstring": "Abstract client responsible for retrieving csv bytes.\n\nRetrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).\n\nNotes:\n **Responsibilities:**\n\n - Implementations must accept a source identifier appropriate to\n the backing store.\n - Return the full csv binary payload.\n - Raise retrieval-specific errors on failure.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.csv.scraper.BaseCsvClient.fetch", - "signature": "", + "signature": "fetch(source: Any)", "docstring": "Fetch raw csv bytes from the given source.\n\nArgs:\n source (Any):\n Identifier of the csv location, such as a file path,\n object storage key, or remote reference.\n\nReturns:\n bytes:\n Raw csv bytes.\n\nRaises:\n Exception:\n Retrieval-specific errors defined by the implementation." } } @@ -557,14 +473,14 @@ "name": "CsvScraper", "kind": "class", "path": "omniread.csv.scraper.CsvScraper", - "signature": "", + "signature": "CsvScraper(*, client: BaseCsvClient)", "docstring": "Scraper for csv documents.\n\nNotes:\n **Responsibilities:**\n\n - Fetch raw csv bytes via the configured client.\n - Wrap the payload in a canonical `Content` instance with the\n CSV content type and source identifier.\n\n **Constraints:**\n\n - The scraper does not perform parsing or interpretation.\n - Does not assume a specific storage backend.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.csv.scraper.CsvScraper.fetch", - "signature": "", + "signature": "fetch(source: Any, *, metadata: Mapping[str, Any] | None = None) -> Content", "docstring": "Fetch a csv document from the given source.\n\nArgs:\n source (Any):\n Identifier of the csv source as understood by the\n configured client.\n metadata (Mapping[str, Any] | None, optional):\n Optional metadata to attach to the returned content.\n\nReturns:\n Content:\n A `Content` instance containing raw csv bytes, source\n identifier, CSV content type, and optional metadata.\n\nRaises:\n Exception:\n Retrieval-specific errors raised by the client." } } diff --git a/mcp/omniread/modules/omniread.csv.parser.json b/mcp/omniread/modules/omniread.csv.parser.json index 1ee8959..50a0fe5 100644 --- a/mcp/omniread/modules/omniread.csv.parser.json +++ b/mcp/omniread/modules/omniread.csv.parser.json @@ -4,60 +4,39 @@ "path": "omniread.csv.parser", "docstring": "# Summary\n\nCSV parser implementations for OmniRead.\n\nThis module provides a concrete, generic parser for comma-separated-value\ndocuments. It exposes records as lists of string cells so downstream\nconsumers can interpret tabular content without depending on the ``csv``\nmodule directly.\n\nThe parser is intentionally statement-agnostic: it performs no header\ndetection or column interpretation beyond basic cell normalization and\ndelimiter detection.", "objects": { - "Sniffer": { - "name": "Sniffer", - "kind": "alias", - "path": "omniread.csv.parser.Sniffer", - "signature": "", - "docstring": null - }, - "reader": { - "name": "reader", - "kind": "alias", - "path": "omniread.csv.parser.reader", - "signature": "", - "docstring": null - }, - "StringIO": { - "name": "StringIO", - "kind": "alias", - "path": "omniread.csv.parser.StringIO", - "signature": "", - "docstring": null - }, "Content": { "name": "Content", "kind": "class", "path": "omniread.csv.parser.Content", - "signature": "", + "signature": "Content(raw: bytes, source: str, content_type: ContentType | None = ..., metadata: Mapping[str, Any] | None = ...)", "docstring": "Normalized representation of extracted content.\n\nNotes:\n **Responsibilities:**\n\n - A `Content` instance represents a raw content payload along with\n minimal contextual metadata describing its origin and type.\n - This class is the primary exchange format between scrapers,\n parsers, and downstream consumers.", "members": { "raw": { "name": "raw", "kind": "attribute", "path": "omniread.csv.parser.Content.raw", - "signature": "", + "signature": null, "docstring": "Raw content bytes as retrieved from the source." }, "source": { "name": "source", "kind": "attribute", "path": "omniread.csv.parser.Content.source", - "signature": "", + "signature": null, "docstring": "Identifier of the content origin (URL, file path, or logical name)." }, "content_type": { "name": "content_type", "kind": "attribute", "path": "omniread.csv.parser.Content.content_type", - "signature": "", + "signature": null, "docstring": "Optional MIME type of the content, if known." }, "metadata": { "name": "metadata", "kind": "attribute", "path": "omniread.csv.parser.Content.metadata", - "signature": "", + "signature": null, "docstring": "Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes)." } } @@ -66,21 +45,21 @@ "name": "CsvParserBase", "kind": "class", "path": "omniread.csv.parser.CsvParserBase", - "signature": "", + "signature": null, "docstring": "Base csv parser.\n\nNotes:\n **Responsibilities:**\n\n - This class enforces csv content-type compatibility and provides\n the extension point for implementing concrete csv parsing\n strategies.\n\n **Constraints:**\n\n - Concrete implementations must define the output type `T` and\n implement the `parse()` method.", "members": { "supported_types": { "name": "supported_types", "kind": "attribute", "path": "omniread.csv.parser.CsvParserBase.supported_types", - "signature": "", + "signature": null, "docstring": "Set of content types supported by this parser (CSV only)." }, "parse": { "name": "parse", "kind": "function", "path": "omniread.csv.parser.CsvParserBase.parse", - "signature": "", + "signature": "parse()", "docstring": "Parse csv content into a structured output.\n\nReturns:\n T:\n Parsed representation of type `T`.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation." } } @@ -89,21 +68,21 @@ "name": "CsvParser", "kind": "class", "path": "omniread.csv.parser.CsvParser", - "signature": "", + "signature": "CsvParser(content: Content)", "docstring": "Generic csv parser producing string rows from the document.\n\nNotes:\n **Responsibilities:**\n\n - Decode the payload (UTF-8 with BOM support, Latin-1 fallback).\n - Detect the delimiter from a leading sample (`,` `;` tab `|`),\n defaulting to `,`.\n - Normalize cells into deterministic stripped string values.\n - Expose row extraction helpers mirroring `XlsxParser.rows`.\n\n **Constraints:**\n\n - All values are strings; consumers requiring typed values must\n convert on their side.\n - Quoted fields containing delimiters/newlines are handled by\n the standard ``csv`` module.", "members": { "parse": { "name": "parse", "kind": "function", "path": "omniread.csv.parser.CsvParser.parse", - "signature": "", + "signature": "parse() -> list[list[str]]", "docstring": "Parse the document into normalized string rows.\n\nReturns:\n list[list[str]]:\n Rows of the document." }, "rows": { "name": "rows", "kind": "function", "path": "omniread.csv.parser.CsvParser.rows", - "signature": "", + "signature": "rows(*, skip_empty: bool = True) -> list[list[str]]", "docstring": "Extract normalized string rows from the document.\n\nArgs:\n skip_empty (bool):\n When True (default), rows whose cells are all blank are\n omitted.\n\nReturns:\n list[list[str]]:\n Normalized rows; trailing blank cells are trimmed per row." } } diff --git a/mcp/omniread/modules/omniread.csv.parser_base.json b/mcp/omniread/modules/omniread.csv.parser_base.json index a0f7802..4c15183 100644 --- a/mcp/omniread/modules/omniread.csv.parser_base.json +++ b/mcp/omniread/modules/omniread.csv.parser_base.json @@ -4,74 +4,53 @@ "path": "omniread.csv.parser_base", "docstring": "# Summary\n\nCSV parser base implementation for OmniRead.\n\nThis module defines the **CSV-specific parser contract**, extending the\nformat-agnostic `BaseParser` with constraints appropriate for\ncomma-separated-value documents.", "objects": { - "abstractmethod": { - "name": "abstractmethod", - "kind": "alias", - "path": "omniread.csv.parser_base.abstractmethod", - "signature": "", - "docstring": null - }, - "Generic": { - "name": "Generic", - "kind": "alias", - "path": "omniread.csv.parser_base.Generic", - "signature": "", - "docstring": null - }, - "TypeVar": { - "name": "TypeVar", - "kind": "alias", - "path": "omniread.csv.parser_base.TypeVar", - "signature": "", - "docstring": null - }, "ContentType": { "name": "ContentType", "kind": "class", "path": "omniread.csv.parser_base.ContentType", - "signature": "", + "signature": null, "docstring": "Supported MIME types for extracted content.\n\nNotes:\n **Guarantees:**\n\n - This enum represents the declared or inferred media type of the\n content source.\n - It is primarily used for routing content to the appropriate\n parser or downstream consumer.", "members": { "HTML": { "name": "HTML", "kind": "attribute", "path": "omniread.csv.parser_base.ContentType.HTML", - "signature": "", + "signature": null, "docstring": "HTML document content." }, "PDF": { "name": "PDF", "kind": "attribute", "path": "omniread.csv.parser_base.ContentType.PDF", - "signature": "", + "signature": null, "docstring": "PDF document content." }, "XLSX": { "name": "XLSX", "kind": "attribute", "path": "omniread.csv.parser_base.ContentType.XLSX", - "signature": "", + "signature": null, "docstring": "Office Open XML spreadsheet (xlsx/xlsm) content." }, "CSV": { "name": "CSV", "kind": "attribute", "path": "omniread.csv.parser_base.ContentType.CSV", - "signature": "", + "signature": null, "docstring": "Comma-separated-value document content." }, "JSON": { "name": "JSON", "kind": "attribute", "path": "omniread.csv.parser_base.ContentType.JSON", - "signature": "", + "signature": null, "docstring": "JSON document content." }, "XML": { "name": "XML", "kind": "attribute", "path": "omniread.csv.parser_base.ContentType.XML", - "signature": "", + "signature": null, "docstring": "XML document content." } } @@ -80,35 +59,35 @@ "name": "BaseParser", "kind": "class", "path": "omniread.csv.parser_base.BaseParser", - "signature": "", + "signature": "BaseParser(content: Content)", "docstring": "Base interface for all parsers.\n\nNotes:\n **Guarantees:**\n\n - A parser is a self-contained object that owns the `Content` it is\n responsible for interpreting.\n - Consumers may rely on early validation of content compatibility\n and type-stable return values from `parse()`.\n\n **Responsibilities:**\n\n - Implementations must declare supported content types via `supported_types`.\n - Implementations must raise parsing-specific exceptions from `parse()`.\n - Implementations must remain deterministic for a given input.", "members": { "supported_types": { "name": "supported_types", "kind": "attribute", "path": "omniread.csv.parser_base.BaseParser.supported_types", - "signature": "", + "signature": null, "docstring": "Set of content types supported by this parser. An empty set indicates that the parser is content-type agnostic." }, "content": { "name": "content", "kind": "attribute", "path": "omniread.csv.parser_base.BaseParser.content", - "signature": "", + "signature": null, "docstring": null }, "parse": { "name": "parse", "kind": "function", "path": "omniread.csv.parser_base.BaseParser.parse", - "signature": "", + "signature": "parse()", "docstring": "Parse the owned content into structured output.\n\nReturns:\n T:\n Parsed, structured representation.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must fully consume the provided content and\n return a deterministic, structured output." }, "supports": { "name": "supports", "kind": "function", "path": "omniread.csv.parser_base.BaseParser.supports", - "signature": "", + "signature": "supports()", "docstring": "Check whether this parser supports the content's type.\n\nReturns:\n bool:\n True if the content type is supported; False otherwise." } } @@ -124,7 +103,7 @@ "name": "CsvParserBase", "kind": "class", "path": "omniread.csv.parser_base.CsvParserBase", - "signature": "", + "signature": null, "docstring": "Base csv parser.\n\nNotes:\n **Responsibilities:**\n\n - This class enforces csv content-type compatibility and provides\n the extension point for implementing concrete csv parsing\n strategies.\n\n **Constraints:**\n\n - Concrete implementations must define the output type `T` and\n implement the `parse()` method.", "members": { "supported_types": { @@ -138,7 +117,7 @@ "name": "parse", "kind": "function", "path": "omniread.csv.parser_base.CsvParserBase.parse", - "signature": "", + "signature": "parse() -> T", "docstring": "Parse csv content into a structured output.\n\nReturns:\n T:\n Parsed representation of type `T`.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation." } } diff --git a/mcp/omniread/modules/omniread.csv.scraper.json b/mcp/omniread/modules/omniread.csv.scraper.json index f1910e9..be51654 100644 --- a/mcp/omniread/modules/omniread.csv.scraper.json +++ b/mcp/omniread/modules/omniread.csv.scraper.json @@ -4,53 +4,39 @@ "path": "omniread.csv.scraper", "docstring": "# Summary\n\nCSV scraper for OmniRead.\n\nThis module defines the scraper responsible for acquiring raw\ncomma-separated-value document content from a backing store via a\nconfigured client.\n\nThe scraper does not interpret or parse the acquired bytes; it wraps them in\nthe canonical `Content` model.", "objects": { - "Mapping": { - "name": "Mapping", - "kind": "alias", - "path": "omniread.csv.scraper.Mapping", - "signature": "", - "docstring": null - }, - "Any": { - "name": "Any", - "kind": "alias", - "path": "omniread.csv.scraper.Any", - "signature": "", - "docstring": null - }, "Content": { "name": "Content", "kind": "class", "path": "omniread.csv.scraper.Content", - "signature": "", + "signature": "Content(raw: bytes, source: str, content_type: ContentType | None = ..., metadata: Mapping[str, Any] | None = ...)", "docstring": "Normalized representation of extracted content.\n\nNotes:\n **Responsibilities:**\n\n - A `Content` instance represents a raw content payload along with\n minimal contextual metadata describing its origin and type.\n - This class is the primary exchange format between scrapers,\n parsers, and downstream consumers.", "members": { "raw": { "name": "raw", "kind": "attribute", "path": "omniread.csv.scraper.Content.raw", - "signature": "", + "signature": null, "docstring": "Raw content bytes as retrieved from the source." }, "source": { "name": "source", "kind": "attribute", "path": "omniread.csv.scraper.Content.source", - "signature": "", + "signature": null, "docstring": "Identifier of the content origin (URL, file path, or logical name)." }, "content_type": { "name": "content_type", "kind": "attribute", "path": "omniread.csv.scraper.Content.content_type", - "signature": "", + "signature": null, "docstring": "Optional MIME type of the content, if known." }, "metadata": { "name": "metadata", "kind": "attribute", "path": "omniread.csv.scraper.Content.metadata", - "signature": "", + "signature": null, "docstring": "Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes)." } } @@ -59,49 +45,49 @@ "name": "ContentType", "kind": "class", "path": "omniread.csv.scraper.ContentType", - "signature": "", + "signature": null, "docstring": "Supported MIME types for extracted content.\n\nNotes:\n **Guarantees:**\n\n - This enum represents the declared or inferred media type of the\n content source.\n - It is primarily used for routing content to the appropriate\n parser or downstream consumer.", "members": { "HTML": { "name": "HTML", "kind": "attribute", "path": "omniread.csv.scraper.ContentType.HTML", - "signature": "", + "signature": null, "docstring": "HTML document content." }, "PDF": { "name": "PDF", "kind": "attribute", "path": "omniread.csv.scraper.ContentType.PDF", - "signature": "", + "signature": null, "docstring": "PDF document content." }, "XLSX": { "name": "XLSX", "kind": "attribute", "path": "omniread.csv.scraper.ContentType.XLSX", - "signature": "", + "signature": null, "docstring": "Office Open XML spreadsheet (xlsx/xlsm) content." }, "CSV": { "name": "CSV", "kind": "attribute", "path": "omniread.csv.scraper.ContentType.CSV", - "signature": "", + "signature": null, "docstring": "Comma-separated-value document content." }, "JSON": { "name": "JSON", "kind": "attribute", "path": "omniread.csv.scraper.ContentType.JSON", - "signature": "", + "signature": null, "docstring": "JSON document content." }, "XML": { "name": "XML", "kind": "attribute", "path": "omniread.csv.scraper.ContentType.XML", - "signature": "", + "signature": null, "docstring": "XML document content." } } @@ -110,14 +96,14 @@ "name": "BaseCsvClient", "kind": "class", "path": "omniread.csv.scraper.BaseCsvClient", - "signature": "", + "signature": null, "docstring": "Abstract client responsible for retrieving csv bytes.\n\nRetrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).\n\nNotes:\n **Responsibilities:**\n\n - Implementations must accept a source identifier appropriate to\n the backing store.\n - Return the full csv binary payload.\n - Raise retrieval-specific errors on failure.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.csv.scraper.BaseCsvClient.fetch", - "signature": "", + "signature": "fetch(source: Any)", "docstring": "Fetch raw csv bytes from the given source.\n\nArgs:\n source (Any):\n Identifier of the csv location, such as a file path,\n object storage key, or remote reference.\n\nReturns:\n bytes:\n Raw csv bytes.\n\nRaises:\n Exception:\n Retrieval-specific errors defined by the implementation." } } @@ -126,14 +112,14 @@ "name": "CsvScraper", "kind": "class", "path": "omniread.csv.scraper.CsvScraper", - "signature": "", + "signature": "CsvScraper(*, client: BaseCsvClient)", "docstring": "Scraper for csv documents.\n\nNotes:\n **Responsibilities:**\n\n - Fetch raw csv bytes via the configured client.\n - Wrap the payload in a canonical `Content` instance with the\n CSV content type and source identifier.\n\n **Constraints:**\n\n - The scraper does not perform parsing or interpretation.\n - Does not assume a specific storage backend.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.csv.scraper.CsvScraper.fetch", - "signature": "", + "signature": "fetch(source: Any, *, metadata: Mapping[str, Any] | None = None) -> Content", "docstring": "Fetch a csv document from the given source.\n\nArgs:\n source (Any):\n Identifier of the csv source as understood by the\n configured client.\n metadata (Mapping[str, Any] | None, optional):\n Optional metadata to attach to the returned content.\n\nReturns:\n Content:\n A `Content` instance containing raw csv bytes, source\n identifier, CSV content type, and optional metadata.\n\nRaises:\n Exception:\n Retrieval-specific errors raised by the client." } } diff --git a/mcp/omniread/modules/omniread.html.json b/mcp/omniread/modules/omniread.html.json index d6ba1d8..3ffb5b5 100644 --- a/mcp/omniread/modules/omniread.html.json +++ b/mcp/omniread/modules/omniread.html.json @@ -8,28 +8,28 @@ "name": "HTMLScraper", "kind": "class", "path": "omniread.html.HTMLScraper", - "signature": "", + "signature": "HTMLScraper(*, client: httpx.Client | None = None, timeout: float = 15.0, headers: Mapping[str, str] | None = None, follow_redirects: bool = True)", "docstring": "Base HTML scraper using `httpx`.\n\nNotes:\n **Responsibilities:**\n\n - This scraper retrieves HTML documents over HTTP(S) and returns\n them as raw content wrapped in a `Content` object.\n - Fetches raw bytes and metadata only.\n - The scraper uses `httpx.Client` for HTTP requests, enforces an\n HTML content type, and preserves HTTP response metadata.\n\n **Constraints:**\n\n - The scraper does not: Parse HTML, perform retries or backoff,\n handle non-HTML responses.", "members": { "content_type": { "name": "content_type", "kind": "attribute", "path": "omniread.html.HTMLScraper.content_type", - "signature": "", + "signature": null, "docstring": null }, "validate_content_type": { "name": "validate_content_type", "kind": "function", "path": "omniread.html.HTMLScraper.validate_content_type", - "signature": "", + "signature": "validate_content_type(response: httpx.Response)", "docstring": "Validate that the HTTP response contains HTML content.\n\nArgs:\n response (httpx.Response):\n HTTP response returned by `httpx`.\n\nRaises:\n ValueError:\n If the `Content-Type` header is missing or does not indicate HTML content." }, "fetch": { "name": "fetch", "kind": "function", "path": "omniread.html.HTMLScraper.fetch", - "signature": "", + "signature": "fetch(source: str, *, metadata: Mapping[str, Any] | None = None)", "docstring": "Fetch an HTML document from the given source.\n\nArgs:\n source (str):\n URL of the HTML document.\n metadata (Mapping[str, Any] | None, optional):\n Optional metadata to be merged into the returned content.\n\nReturns:\n Content:\n A `Content` instance containing raw HTML bytes, source URL, HTML content type, and HTTP response metadata.\n\nRaises:\n httpx.HTTPError:\n If the HTTP request fails.\n ValueError:\n If the response is not valid HTML." } } @@ -38,49 +38,49 @@ "name": "HTMLParser", "kind": "class", "path": "omniread.html.HTMLParser", - "signature": "", + "signature": "HTMLParser(content: Content, features: str = 'html.parser')", "docstring": "Base HTML parser.\n\nNotes:\n **Responsibilities:**\n\n - This class extends the core `BaseParser` with HTML-specific behavior,\n including DOM parsing via BeautifulSoup and reusable extraction helpers.\n - Provides reusable helpers for HTML extraction. Concrete parsers must\n explicitly define the return type.\n\n **Guarantees:**\n\n - Accepts only HTML content.\n - Owns a parsed BeautifulSoup DOM tree.\n - Provides pure helper utilities for common HTML structures.\n\n **Constraints:**\n\n - Concrete subclasses must define the output type `T` and implement\n the `parse()` method.", "members": { "supported_types": { "name": "supported_types", "kind": "attribute", "path": "omniread.html.HTMLParser.supported_types", - "signature": "", + "signature": null, "docstring": "Set of content types supported by this parser (HTML only)." }, "parse": { "name": "parse", "kind": "function", "path": "omniread.html.HTMLParser.parse", - "signature": "", + "signature": "parse()", "docstring": "Fully parse the HTML content into structured output.\n\nReturns:\n T:\n Parsed representation of type `T`.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must fully interpret the HTML DOM and return a\n deterministic, structured output." }, "parse_div": { "name": "parse_div", "kind": "function", "path": "omniread.html.HTMLParser.parse_div", - "signature": "", + "signature": "parse_div(div: Tag, *, separator: str = ' ')", "docstring": "Extract normalized text from a `
` element.\n\nArgs:\n div (Tag):\n BeautifulSoup tag representing a `
`.\n separator (str, optional):\n String used to separate text nodes.\n\nReturns:\n str:\n Flattened, whitespace-normalized text content." }, "parse_link": { "name": "parse_link", "kind": "function", "path": "omniread.html.HTMLParser.parse_link", - "signature": "", + "signature": "parse_link(a: Tag)", "docstring": "Extract the hyperlink reference from an `` element.\n\nArgs:\n a (Tag):\n BeautifulSoup tag representing an anchor.\n\nReturns:\n str | None:\n The value of the `href` attribute, or None if absent." }, "parse_table": { "name": "parse_table", "kind": "function", "path": "omniread.html.HTMLParser.parse_table", - "signature": "", + "signature": "parse_table(table: Tag)", "docstring": "Parse an HTML table into a 2D list of strings.\n\nArgs:\n table (Tag):\n BeautifulSoup tag representing a ``.\n\nReturns:\n list[list[str]]:\n A list of rows, where each row is a list of cell text values." }, "parse_meta": { "name": "parse_meta", "kind": "function", "path": "omniread.html.HTMLParser.parse_meta", - "signature": "", + "signature": "parse_meta()", "docstring": "Extract high-level metadata from the HTML document.\n\nReturns:\n dict[str, Any]:\n Dictionary containing extracted metadata.\n\nNotes:\n **Responsibilities:**\n\n - Extract high-level metadata from the HTML document.\n - This includes: Document title, `` tag name/property to\n content mappings." } } @@ -92,81 +92,39 @@ "signature": null, "docstring": "# Summary\n\nHTML parser base implementations for OmniRead.\n\nThis module provides reusable HTML parsing utilities built on top of\nthe abstract parser contracts defined in `omniread.core.parser`.\n\nIt supplies:\n\n- Content-type enforcement for HTML inputs\n- BeautifulSoup initialization and lifecycle management\n- Common helper methods for extracting structured data from HTML elements\n\nConcrete parsers must subclass `HTMLParser` and implement the `parse()` method\nto return a structured representation appropriate for their use case.", "members": { - "abstractmethod": { - "name": "abstractmethod", - "kind": "alias", - "path": "omniread.html.parser.abstractmethod", - "signature": "", - "docstring": null - }, - "Any": { - "name": "Any", - "kind": "alias", - "path": "omniread.html.parser.Any", - "signature": "", - "docstring": null - }, - "Generic": { - "name": "Generic", - "kind": "alias", - "path": "omniread.html.parser.Generic", - "signature": "", - "docstring": null - }, - "TypeVar": { - "name": "TypeVar", - "kind": "alias", - "path": "omniread.html.parser.TypeVar", - "signature": "", - "docstring": null - }, - "BeautifulSoup": { - "name": "BeautifulSoup", - "kind": "alias", - "path": "omniread.html.parser.BeautifulSoup", - "signature": "", - "docstring": null - }, - "Tag": { - "name": "Tag", - "kind": "alias", - "path": "omniread.html.parser.Tag", - "signature": "", - "docstring": null - }, "Content": { "name": "Content", "kind": "class", "path": "omniread.html.parser.Content", - "signature": "", + "signature": "Content(raw: bytes, source: str, content_type: ContentType | None = ..., metadata: Mapping[str, Any] | None = ...)", "docstring": "Normalized representation of extracted content.\n\nNotes:\n **Responsibilities:**\n\n - A `Content` instance represents a raw content payload along with\n minimal contextual metadata describing its origin and type.\n - This class is the primary exchange format between scrapers,\n parsers, and downstream consumers.", "members": { "raw": { "name": "raw", "kind": "attribute", "path": "omniread.html.parser.Content.raw", - "signature": "", + "signature": null, "docstring": "Raw content bytes as retrieved from the source." }, "source": { "name": "source", "kind": "attribute", "path": "omniread.html.parser.Content.source", - "signature": "", + "signature": null, "docstring": "Identifier of the content origin (URL, file path, or logical name)." }, "content_type": { "name": "content_type", "kind": "attribute", "path": "omniread.html.parser.Content.content_type", - "signature": "", + "signature": null, "docstring": "Optional MIME type of the content, if known." }, "metadata": { "name": "metadata", "kind": "attribute", "path": "omniread.html.parser.Content.metadata", - "signature": "", + "signature": null, "docstring": "Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes)." } } @@ -175,49 +133,49 @@ "name": "ContentType", "kind": "class", "path": "omniread.html.parser.ContentType", - "signature": "", + "signature": null, "docstring": "Supported MIME types for extracted content.\n\nNotes:\n **Guarantees:**\n\n - This enum represents the declared or inferred media type of the\n content source.\n - It is primarily used for routing content to the appropriate\n parser or downstream consumer.", "members": { "HTML": { "name": "HTML", "kind": "attribute", "path": "omniread.html.parser.ContentType.HTML", - "signature": "", + "signature": null, "docstring": "HTML document content." }, "PDF": { "name": "PDF", "kind": "attribute", "path": "omniread.html.parser.ContentType.PDF", - "signature": "", + "signature": null, "docstring": "PDF document content." }, "XLSX": { "name": "XLSX", "kind": "attribute", "path": "omniread.html.parser.ContentType.XLSX", - "signature": "", + "signature": null, "docstring": "Office Open XML spreadsheet (xlsx/xlsm) content." }, "CSV": { "name": "CSV", "kind": "attribute", "path": "omniread.html.parser.ContentType.CSV", - "signature": "", + "signature": null, "docstring": "Comma-separated-value document content." }, "JSON": { "name": "JSON", "kind": "attribute", "path": "omniread.html.parser.ContentType.JSON", - "signature": "", + "signature": null, "docstring": "JSON document content." }, "XML": { "name": "XML", "kind": "attribute", "path": "omniread.html.parser.ContentType.XML", - "signature": "", + "signature": null, "docstring": "XML document content." } } @@ -226,35 +184,35 @@ "name": "BaseParser", "kind": "class", "path": "omniread.html.parser.BaseParser", - "signature": "", + "signature": "BaseParser(content: Content)", "docstring": "Base interface for all parsers.\n\nNotes:\n **Guarantees:**\n\n - A parser is a self-contained object that owns the `Content` it is\n responsible for interpreting.\n - Consumers may rely on early validation of content compatibility\n and type-stable return values from `parse()`.\n\n **Responsibilities:**\n\n - Implementations must declare supported content types via `supported_types`.\n - Implementations must raise parsing-specific exceptions from `parse()`.\n - Implementations must remain deterministic for a given input.", "members": { "supported_types": { "name": "supported_types", "kind": "attribute", "path": "omniread.html.parser.BaseParser.supported_types", - "signature": "", + "signature": null, "docstring": "Set of content types supported by this parser. An empty set indicates that the parser is content-type agnostic." }, "content": { "name": "content", "kind": "attribute", "path": "omniread.html.parser.BaseParser.content", - "signature": "", + "signature": null, "docstring": null }, "parse": { "name": "parse", "kind": "function", "path": "omniread.html.parser.BaseParser.parse", - "signature": "", + "signature": "parse()", "docstring": "Parse the owned content into structured output.\n\nReturns:\n T:\n Parsed, structured representation.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must fully consume the provided content and\n return a deterministic, structured output." }, "supports": { "name": "supports", "kind": "function", "path": "omniread.html.parser.BaseParser.supports", - "signature": "", + "signature": "supports()", "docstring": "Check whether this parser supports the content's type.\n\nReturns:\n bool:\n True if the content type is supported; False otherwise." } } @@ -270,7 +228,7 @@ "name": "HTMLParser", "kind": "class", "path": "omniread.html.parser.HTMLParser", - "signature": "", + "signature": "HTMLParser(content: Content, features: str = 'html.parser')", "docstring": "Base HTML parser.\n\nNotes:\n **Responsibilities:**\n\n - This class extends the core `BaseParser` with HTML-specific behavior,\n including DOM parsing via BeautifulSoup and reusable extraction helpers.\n - Provides reusable helpers for HTML extraction. Concrete parsers must\n explicitly define the return type.\n\n **Guarantees:**\n\n - Accepts only HTML content.\n - Owns a parsed BeautifulSoup DOM tree.\n - Provides pure helper utilities for common HTML structures.\n\n **Constraints:**\n\n - Concrete subclasses must define the output type `T` and implement\n the `parse()` method.", "members": { "supported_types": { @@ -284,35 +242,35 @@ "name": "parse", "kind": "function", "path": "omniread.html.parser.HTMLParser.parse", - "signature": "", + "signature": "parse() -> T", "docstring": "Fully parse the HTML content into structured output.\n\nReturns:\n T:\n Parsed representation of type `T`.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must fully interpret the HTML DOM and return a\n deterministic, structured output." }, "parse_div": { "name": "parse_div", "kind": "function", "path": "omniread.html.parser.HTMLParser.parse_div", - "signature": "", + "signature": "parse_div(div: Tag, *, separator: str = ' ') -> str", "docstring": "Extract normalized text from a `
` element.\n\nArgs:\n div (Tag):\n BeautifulSoup tag representing a `
`.\n separator (str, optional):\n String used to separate text nodes.\n\nReturns:\n str:\n Flattened, whitespace-normalized text content." }, "parse_link": { "name": "parse_link", "kind": "function", "path": "omniread.html.parser.HTMLParser.parse_link", - "signature": "", + "signature": "parse_link(a: Tag) -> str | None", "docstring": "Extract the hyperlink reference from an `` element.\n\nArgs:\n a (Tag):\n BeautifulSoup tag representing an anchor.\n\nReturns:\n str | None:\n The value of the `href` attribute, or None if absent." }, "parse_table": { "name": "parse_table", "kind": "function", "path": "omniread.html.parser.HTMLParser.parse_table", - "signature": "", + "signature": "parse_table(table: Tag) -> list[list[str]]", "docstring": "Parse an HTML table into a 2D list of strings.\n\nArgs:\n table (Tag):\n BeautifulSoup tag representing a `
`.\n\nReturns:\n list[list[str]]:\n A list of rows, where each row is a list of cell text values." }, "parse_meta": { "name": "parse_meta", "kind": "function", "path": "omniread.html.parser.HTMLParser.parse_meta", - "signature": "", + "signature": "parse_meta() -> dict[str, Any]", "docstring": "Extract high-level metadata from the HTML document.\n\nReturns:\n dict[str, Any]:\n Dictionary containing extracted metadata.\n\nNotes:\n **Responsibilities:**\n\n - Extract high-level metadata from the HTML document.\n - This includes: Document title, `` tag name/property to\n content mappings." } } @@ -326,60 +284,39 @@ "signature": null, "docstring": "# Summary\n\nHTML scraping implementation for OmniRead.\n\nThis module provides an HTTP-based scraper for retrieving HTML documents.\nIt implements the core `BaseScraper` contract using `httpx` as the transport\nlayer.\n\nThis scraper is responsible for:\n\n- Fetching raw HTML bytes over HTTP(S)\n- Validating response content type\n- Attaching HTTP metadata to the returned content\n\nThis scraper is not responsible for:\n\n- Parsing or interpreting HTML\n- Retrying failed requests\n- Managing crawl policies or rate limiting", "members": { - "Mapping": { - "name": "Mapping", - "kind": "alias", - "path": "omniread.html.scraper.Mapping", - "signature": "", - "docstring": null - }, - "Any": { - "name": "Any", - "kind": "alias", - "path": "omniread.html.scraper.Any", - "signature": "", - "docstring": null - }, - "httpx": { - "name": "httpx", - "kind": "alias", - "path": "omniread.html.scraper.httpx", - "signature": "", - "docstring": null - }, "Content": { "name": "Content", "kind": "class", "path": "omniread.html.scraper.Content", - "signature": "", + "signature": "Content(raw: bytes, source: str, content_type: ContentType | None = ..., metadata: Mapping[str, Any] | None = ...)", "docstring": "Normalized representation of extracted content.\n\nNotes:\n **Responsibilities:**\n\n - A `Content` instance represents a raw content payload along with\n minimal contextual metadata describing its origin and type.\n - This class is the primary exchange format between scrapers,\n parsers, and downstream consumers.", "members": { "raw": { "name": "raw", "kind": "attribute", "path": "omniread.html.scraper.Content.raw", - "signature": "", + "signature": null, "docstring": "Raw content bytes as retrieved from the source." }, "source": { "name": "source", "kind": "attribute", "path": "omniread.html.scraper.Content.source", - "signature": "", + "signature": null, "docstring": "Identifier of the content origin (URL, file path, or logical name)." }, "content_type": { "name": "content_type", "kind": "attribute", "path": "omniread.html.scraper.Content.content_type", - "signature": "", + "signature": null, "docstring": "Optional MIME type of the content, if known." }, "metadata": { "name": "metadata", "kind": "attribute", "path": "omniread.html.scraper.Content.metadata", - "signature": "", + "signature": null, "docstring": "Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes)." } } @@ -388,49 +325,49 @@ "name": "ContentType", "kind": "class", "path": "omniread.html.scraper.ContentType", - "signature": "", + "signature": null, "docstring": "Supported MIME types for extracted content.\n\nNotes:\n **Guarantees:**\n\n - This enum represents the declared or inferred media type of the\n content source.\n - It is primarily used for routing content to the appropriate\n parser or downstream consumer.", "members": { "HTML": { "name": "HTML", "kind": "attribute", "path": "omniread.html.scraper.ContentType.HTML", - "signature": "", + "signature": null, "docstring": "HTML document content." }, "PDF": { "name": "PDF", "kind": "attribute", "path": "omniread.html.scraper.ContentType.PDF", - "signature": "", + "signature": null, "docstring": "PDF document content." }, "XLSX": { "name": "XLSX", "kind": "attribute", "path": "omniread.html.scraper.ContentType.XLSX", - "signature": "", + "signature": null, "docstring": "Office Open XML spreadsheet (xlsx/xlsm) content." }, "CSV": { "name": "CSV", "kind": "attribute", "path": "omniread.html.scraper.ContentType.CSV", - "signature": "", + "signature": null, "docstring": "Comma-separated-value document content." }, "JSON": { "name": "JSON", "kind": "attribute", "path": "omniread.html.scraper.ContentType.JSON", - "signature": "", + "signature": null, "docstring": "JSON document content." }, "XML": { "name": "XML", "kind": "attribute", "path": "omniread.html.scraper.ContentType.XML", - "signature": "", + "signature": null, "docstring": "XML document content." } } @@ -439,14 +376,14 @@ "name": "BaseScraper", "kind": "class", "path": "omniread.html.scraper.BaseScraper", - "signature": "", + "signature": null, "docstring": "Base interface for all scrapers.\n\nNotes:\n **Responsibilities:**\n\n - A scraper is responsible ONLY for fetching raw content (bytes)\n from a source. It must not interpret or parse it.\n - A scraper is a stateless acquisition component that retrieves raw\n content from a source and returns it as a `Content` object.\n - Scrapers define how content is obtained, not what the content means.\n - Implementations may vary in transport mechanism, authentication\n strategy, retry and backoff behavior.\n\n **Constraints:**\n\n - Implementations must not parse content, modify content semantics,\n or couple scraping logic to a specific parser.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.html.scraper.BaseScraper.fetch", - "signature": "", + "signature": "fetch(source: str, *, metadata: Mapping[str, Any] | None = None)", "docstring": "Fetch raw content from the given source.\n\nArgs:\n source (str):\n Location identifier (URL, file path, S3 URI, etc.).\n\n metadata (Mapping[str, Any] | None, optional):\n Optional hints for the scraper (headers, auth, etc.).\n\nReturns:\n Content:\n Content object containing raw bytes and metadata.\n\nRaises:\n Exception:\n Retrieval-specific errors as defined by the implementation.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must retrieve the content referenced by `source`\n and return it as raw bytes wrapped in a `Content` object." } } @@ -455,7 +392,7 @@ "name": "HTMLScraper", "kind": "class", "path": "omniread.html.scraper.HTMLScraper", - "signature": "", + "signature": "HTMLScraper(*, client: httpx.Client | None = None, timeout: float = 15.0, headers: Mapping[str, str] | None = None, follow_redirects: bool = True)", "docstring": "Base HTML scraper using `httpx`.\n\nNotes:\n **Responsibilities:**\n\n - This scraper retrieves HTML documents over HTTP(S) and returns\n them as raw content wrapped in a `Content` object.\n - Fetches raw bytes and metadata only.\n - The scraper uses `httpx.Client` for HTTP requests, enforces an\n HTML content type, and preserves HTTP response metadata.\n\n **Constraints:**\n\n - The scraper does not: Parse HTML, perform retries or backoff,\n handle non-HTML responses.", "members": { "content_type": { @@ -469,14 +406,14 @@ "name": "validate_content_type", "kind": "function", "path": "omniread.html.scraper.HTMLScraper.validate_content_type", - "signature": "", + "signature": "validate_content_type(response: httpx.Response) -> None", "docstring": "Validate that the HTTP response contains HTML content.\n\nArgs:\n response (httpx.Response):\n HTTP response returned by `httpx`.\n\nRaises:\n ValueError:\n If the `Content-Type` header is missing or does not indicate HTML content." }, "fetch": { "name": "fetch", "kind": "function", "path": "omniread.html.scraper.HTMLScraper.fetch", - "signature": "", + "signature": "fetch(source: str, *, metadata: Mapping[str, Any] | None = None) -> Content", "docstring": "Fetch an HTML document from the given source.\n\nArgs:\n source (str):\n URL of the HTML document.\n metadata (Mapping[str, Any] | None, optional):\n Optional metadata to be merged into the returned content.\n\nReturns:\n Content:\n A `Content` instance containing raw HTML bytes, source URL, HTML content type, and HTTP response metadata.\n\nRaises:\n httpx.HTTPError:\n If the HTTP request fails.\n ValueError:\n If the response is not valid HTML." } } diff --git a/mcp/omniread/modules/omniread.html.parser.json b/mcp/omniread/modules/omniread.html.parser.json index 15ca5c5..31c9c59 100644 --- a/mcp/omniread/modules/omniread.html.parser.json +++ b/mcp/omniread/modules/omniread.html.parser.json @@ -4,81 +4,39 @@ "path": "omniread.html.parser", "docstring": "# Summary\n\nHTML parser base implementations for OmniRead.\n\nThis module provides reusable HTML parsing utilities built on top of\nthe abstract parser contracts defined in `omniread.core.parser`.\n\nIt supplies:\n\n- Content-type enforcement for HTML inputs\n- BeautifulSoup initialization and lifecycle management\n- Common helper methods for extracting structured data from HTML elements\n\nConcrete parsers must subclass `HTMLParser` and implement the `parse()` method\nto return a structured representation appropriate for their use case.", "objects": { - "abstractmethod": { - "name": "abstractmethod", - "kind": "alias", - "path": "omniread.html.parser.abstractmethod", - "signature": "", - "docstring": null - }, - "Any": { - "name": "Any", - "kind": "alias", - "path": "omniread.html.parser.Any", - "signature": "", - "docstring": null - }, - "Generic": { - "name": "Generic", - "kind": "alias", - "path": "omniread.html.parser.Generic", - "signature": "", - "docstring": null - }, - "TypeVar": { - "name": "TypeVar", - "kind": "alias", - "path": "omniread.html.parser.TypeVar", - "signature": "", - "docstring": null - }, - "BeautifulSoup": { - "name": "BeautifulSoup", - "kind": "alias", - "path": "omniread.html.parser.BeautifulSoup", - "signature": "", - "docstring": null - }, - "Tag": { - "name": "Tag", - "kind": "alias", - "path": "omniread.html.parser.Tag", - "signature": "", - "docstring": null - }, "Content": { "name": "Content", "kind": "class", "path": "omniread.html.parser.Content", - "signature": "", + "signature": "Content(raw: bytes, source: str, content_type: ContentType | None = ..., metadata: Mapping[str, Any] | None = ...)", "docstring": "Normalized representation of extracted content.\n\nNotes:\n **Responsibilities:**\n\n - A `Content` instance represents a raw content payload along with\n minimal contextual metadata describing its origin and type.\n - This class is the primary exchange format between scrapers,\n parsers, and downstream consumers.", "members": { "raw": { "name": "raw", "kind": "attribute", "path": "omniread.html.parser.Content.raw", - "signature": "", + "signature": null, "docstring": "Raw content bytes as retrieved from the source." }, "source": { "name": "source", "kind": "attribute", "path": "omniread.html.parser.Content.source", - "signature": "", + "signature": null, "docstring": "Identifier of the content origin (URL, file path, or logical name)." }, "content_type": { "name": "content_type", "kind": "attribute", "path": "omniread.html.parser.Content.content_type", - "signature": "", + "signature": null, "docstring": "Optional MIME type of the content, if known." }, "metadata": { "name": "metadata", "kind": "attribute", "path": "omniread.html.parser.Content.metadata", - "signature": "", + "signature": null, "docstring": "Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes)." } } @@ -87,49 +45,49 @@ "name": "ContentType", "kind": "class", "path": "omniread.html.parser.ContentType", - "signature": "", + "signature": null, "docstring": "Supported MIME types for extracted content.\n\nNotes:\n **Guarantees:**\n\n - This enum represents the declared or inferred media type of the\n content source.\n - It is primarily used for routing content to the appropriate\n parser or downstream consumer.", "members": { "HTML": { "name": "HTML", "kind": "attribute", "path": "omniread.html.parser.ContentType.HTML", - "signature": "", + "signature": null, "docstring": "HTML document content." }, "PDF": { "name": "PDF", "kind": "attribute", "path": "omniread.html.parser.ContentType.PDF", - "signature": "", + "signature": null, "docstring": "PDF document content." }, "XLSX": { "name": "XLSX", "kind": "attribute", "path": "omniread.html.parser.ContentType.XLSX", - "signature": "", + "signature": null, "docstring": "Office Open XML spreadsheet (xlsx/xlsm) content." }, "CSV": { "name": "CSV", "kind": "attribute", "path": "omniread.html.parser.ContentType.CSV", - "signature": "", + "signature": null, "docstring": "Comma-separated-value document content." }, "JSON": { "name": "JSON", "kind": "attribute", "path": "omniread.html.parser.ContentType.JSON", - "signature": "", + "signature": null, "docstring": "JSON document content." }, "XML": { "name": "XML", "kind": "attribute", "path": "omniread.html.parser.ContentType.XML", - "signature": "", + "signature": null, "docstring": "XML document content." } } @@ -138,35 +96,35 @@ "name": "BaseParser", "kind": "class", "path": "omniread.html.parser.BaseParser", - "signature": "", + "signature": "BaseParser(content: Content)", "docstring": "Base interface for all parsers.\n\nNotes:\n **Guarantees:**\n\n - A parser is a self-contained object that owns the `Content` it is\n responsible for interpreting.\n - Consumers may rely on early validation of content compatibility\n and type-stable return values from `parse()`.\n\n **Responsibilities:**\n\n - Implementations must declare supported content types via `supported_types`.\n - Implementations must raise parsing-specific exceptions from `parse()`.\n - Implementations must remain deterministic for a given input.", "members": { "supported_types": { "name": "supported_types", "kind": "attribute", "path": "omniread.html.parser.BaseParser.supported_types", - "signature": "", + "signature": null, "docstring": "Set of content types supported by this parser. An empty set indicates that the parser is content-type agnostic." }, "content": { "name": "content", "kind": "attribute", "path": "omniread.html.parser.BaseParser.content", - "signature": "", + "signature": null, "docstring": null }, "parse": { "name": "parse", "kind": "function", "path": "omniread.html.parser.BaseParser.parse", - "signature": "", + "signature": "parse()", "docstring": "Parse the owned content into structured output.\n\nReturns:\n T:\n Parsed, structured representation.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must fully consume the provided content and\n return a deterministic, structured output." }, "supports": { "name": "supports", "kind": "function", "path": "omniread.html.parser.BaseParser.supports", - "signature": "", + "signature": "supports()", "docstring": "Check whether this parser supports the content's type.\n\nReturns:\n bool:\n True if the content type is supported; False otherwise." } } @@ -182,7 +140,7 @@ "name": "HTMLParser", "kind": "class", "path": "omniread.html.parser.HTMLParser", - "signature": "", + "signature": "HTMLParser(content: Content, features: str = 'html.parser')", "docstring": "Base HTML parser.\n\nNotes:\n **Responsibilities:**\n\n - This class extends the core `BaseParser` with HTML-specific behavior,\n including DOM parsing via BeautifulSoup and reusable extraction helpers.\n - Provides reusable helpers for HTML extraction. Concrete parsers must\n explicitly define the return type.\n\n **Guarantees:**\n\n - Accepts only HTML content.\n - Owns a parsed BeautifulSoup DOM tree.\n - Provides pure helper utilities for common HTML structures.\n\n **Constraints:**\n\n - Concrete subclasses must define the output type `T` and implement\n the `parse()` method.", "members": { "supported_types": { @@ -196,35 +154,35 @@ "name": "parse", "kind": "function", "path": "omniread.html.parser.HTMLParser.parse", - "signature": "", + "signature": "parse() -> T", "docstring": "Fully parse the HTML content into structured output.\n\nReturns:\n T:\n Parsed representation of type `T`.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must fully interpret the HTML DOM and return a\n deterministic, structured output." }, "parse_div": { "name": "parse_div", "kind": "function", "path": "omniread.html.parser.HTMLParser.parse_div", - "signature": "", + "signature": "parse_div(div: Tag, *, separator: str = ' ') -> str", "docstring": "Extract normalized text from a `
` element.\n\nArgs:\n div (Tag):\n BeautifulSoup tag representing a `
`.\n separator (str, optional):\n String used to separate text nodes.\n\nReturns:\n str:\n Flattened, whitespace-normalized text content." }, "parse_link": { "name": "parse_link", "kind": "function", "path": "omniread.html.parser.HTMLParser.parse_link", - "signature": "", + "signature": "parse_link(a: Tag) -> str | None", "docstring": "Extract the hyperlink reference from an `` element.\n\nArgs:\n a (Tag):\n BeautifulSoup tag representing an anchor.\n\nReturns:\n str | None:\n The value of the `href` attribute, or None if absent." }, "parse_table": { "name": "parse_table", "kind": "function", "path": "omniread.html.parser.HTMLParser.parse_table", - "signature": "", + "signature": "parse_table(table: Tag) -> list[list[str]]", "docstring": "Parse an HTML table into a 2D list of strings.\n\nArgs:\n table (Tag):\n BeautifulSoup tag representing a `
`.\n\nReturns:\n list[list[str]]:\n A list of rows, where each row is a list of cell text values." }, "parse_meta": { "name": "parse_meta", "kind": "function", "path": "omniread.html.parser.HTMLParser.parse_meta", - "signature": "", + "signature": "parse_meta() -> dict[str, Any]", "docstring": "Extract high-level metadata from the HTML document.\n\nReturns:\n dict[str, Any]:\n Dictionary containing extracted metadata.\n\nNotes:\n **Responsibilities:**\n\n - Extract high-level metadata from the HTML document.\n - This includes: Document title, `` tag name/property to\n content mappings." } } diff --git a/mcp/omniread/modules/omniread.html.scraper.json b/mcp/omniread/modules/omniread.html.scraper.json index 7f15d29..2eda9e9 100644 --- a/mcp/omniread/modules/omniread.html.scraper.json +++ b/mcp/omniread/modules/omniread.html.scraper.json @@ -4,60 +4,39 @@ "path": "omniread.html.scraper", "docstring": "# Summary\n\nHTML scraping implementation for OmniRead.\n\nThis module provides an HTTP-based scraper for retrieving HTML documents.\nIt implements the core `BaseScraper` contract using `httpx` as the transport\nlayer.\n\nThis scraper is responsible for:\n\n- Fetching raw HTML bytes over HTTP(S)\n- Validating response content type\n- Attaching HTTP metadata to the returned content\n\nThis scraper is not responsible for:\n\n- Parsing or interpreting HTML\n- Retrying failed requests\n- Managing crawl policies or rate limiting", "objects": { - "Mapping": { - "name": "Mapping", - "kind": "alias", - "path": "omniread.html.scraper.Mapping", - "signature": "", - "docstring": null - }, - "Any": { - "name": "Any", - "kind": "alias", - "path": "omniread.html.scraper.Any", - "signature": "", - "docstring": null - }, - "httpx": { - "name": "httpx", - "kind": "alias", - "path": "omniread.html.scraper.httpx", - "signature": "", - "docstring": null - }, "Content": { "name": "Content", "kind": "class", "path": "omniread.html.scraper.Content", - "signature": "", + "signature": "Content(raw: bytes, source: str, content_type: ContentType | None = ..., metadata: Mapping[str, Any] | None = ...)", "docstring": "Normalized representation of extracted content.\n\nNotes:\n **Responsibilities:**\n\n - A `Content` instance represents a raw content payload along with\n minimal contextual metadata describing its origin and type.\n - This class is the primary exchange format between scrapers,\n parsers, and downstream consumers.", "members": { "raw": { "name": "raw", "kind": "attribute", "path": "omniread.html.scraper.Content.raw", - "signature": "", + "signature": null, "docstring": "Raw content bytes as retrieved from the source." }, "source": { "name": "source", "kind": "attribute", "path": "omniread.html.scraper.Content.source", - "signature": "", + "signature": null, "docstring": "Identifier of the content origin (URL, file path, or logical name)." }, "content_type": { "name": "content_type", "kind": "attribute", "path": "omniread.html.scraper.Content.content_type", - "signature": "", + "signature": null, "docstring": "Optional MIME type of the content, if known." }, "metadata": { "name": "metadata", "kind": "attribute", "path": "omniread.html.scraper.Content.metadata", - "signature": "", + "signature": null, "docstring": "Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes)." } } @@ -66,49 +45,49 @@ "name": "ContentType", "kind": "class", "path": "omniread.html.scraper.ContentType", - "signature": "", + "signature": null, "docstring": "Supported MIME types for extracted content.\n\nNotes:\n **Guarantees:**\n\n - This enum represents the declared or inferred media type of the\n content source.\n - It is primarily used for routing content to the appropriate\n parser or downstream consumer.", "members": { "HTML": { "name": "HTML", "kind": "attribute", "path": "omniread.html.scraper.ContentType.HTML", - "signature": "", + "signature": null, "docstring": "HTML document content." }, "PDF": { "name": "PDF", "kind": "attribute", "path": "omniread.html.scraper.ContentType.PDF", - "signature": "", + "signature": null, "docstring": "PDF document content." }, "XLSX": { "name": "XLSX", "kind": "attribute", "path": "omniread.html.scraper.ContentType.XLSX", - "signature": "", + "signature": null, "docstring": "Office Open XML spreadsheet (xlsx/xlsm) content." }, "CSV": { "name": "CSV", "kind": "attribute", "path": "omniread.html.scraper.ContentType.CSV", - "signature": "", + "signature": null, "docstring": "Comma-separated-value document content." }, "JSON": { "name": "JSON", "kind": "attribute", "path": "omniread.html.scraper.ContentType.JSON", - "signature": "", + "signature": null, "docstring": "JSON document content." }, "XML": { "name": "XML", "kind": "attribute", "path": "omniread.html.scraper.ContentType.XML", - "signature": "", + "signature": null, "docstring": "XML document content." } } @@ -117,14 +96,14 @@ "name": "BaseScraper", "kind": "class", "path": "omniread.html.scraper.BaseScraper", - "signature": "", + "signature": null, "docstring": "Base interface for all scrapers.\n\nNotes:\n **Responsibilities:**\n\n - A scraper is responsible ONLY for fetching raw content (bytes)\n from a source. It must not interpret or parse it.\n - A scraper is a stateless acquisition component that retrieves raw\n content from a source and returns it as a `Content` object.\n - Scrapers define how content is obtained, not what the content means.\n - Implementations may vary in transport mechanism, authentication\n strategy, retry and backoff behavior.\n\n **Constraints:**\n\n - Implementations must not parse content, modify content semantics,\n or couple scraping logic to a specific parser.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.html.scraper.BaseScraper.fetch", - "signature": "", + "signature": "fetch(source: str, *, metadata: Mapping[str, Any] | None = None)", "docstring": "Fetch raw content from the given source.\n\nArgs:\n source (str):\n Location identifier (URL, file path, S3 URI, etc.).\n\n metadata (Mapping[str, Any] | None, optional):\n Optional hints for the scraper (headers, auth, etc.).\n\nReturns:\n Content:\n Content object containing raw bytes and metadata.\n\nRaises:\n Exception:\n Retrieval-specific errors as defined by the implementation.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must retrieve the content referenced by `source`\n and return it as raw bytes wrapped in a `Content` object." } } @@ -133,7 +112,7 @@ "name": "HTMLScraper", "kind": "class", "path": "omniread.html.scraper.HTMLScraper", - "signature": "", + "signature": "HTMLScraper(*, client: httpx.Client | None = None, timeout: float = 15.0, headers: Mapping[str, str] | None = None, follow_redirects: bool = True)", "docstring": "Base HTML scraper using `httpx`.\n\nNotes:\n **Responsibilities:**\n\n - This scraper retrieves HTML documents over HTTP(S) and returns\n them as raw content wrapped in a `Content` object.\n - Fetches raw bytes and metadata only.\n - The scraper uses `httpx.Client` for HTTP requests, enforces an\n HTML content type, and preserves HTTP response metadata.\n\n **Constraints:**\n\n - The scraper does not: Parse HTML, perform retries or backoff,\n handle non-HTML responses.", "members": { "content_type": { @@ -147,14 +126,14 @@ "name": "validate_content_type", "kind": "function", "path": "omniread.html.scraper.HTMLScraper.validate_content_type", - "signature": "", + "signature": "validate_content_type(response: httpx.Response) -> None", "docstring": "Validate that the HTTP response contains HTML content.\n\nArgs:\n response (httpx.Response):\n HTTP response returned by `httpx`.\n\nRaises:\n ValueError:\n If the `Content-Type` header is missing or does not indicate HTML content." }, "fetch": { "name": "fetch", "kind": "function", "path": "omniread.html.scraper.HTMLScraper.fetch", - "signature": "", + "signature": "fetch(source: str, *, metadata: Mapping[str, Any] | None = None) -> Content", "docstring": "Fetch an HTML document from the given source.\n\nArgs:\n source (str):\n URL of the HTML document.\n metadata (Mapping[str, Any] | None, optional):\n Optional metadata to be merged into the returned content.\n\nReturns:\n Content:\n A `Content` instance containing raw HTML bytes, source URL, HTML content type, and HTTP response metadata.\n\nRaises:\n httpx.HTTPError:\n If the HTTP request fails.\n ValueError:\n If the response is not valid HTML." } } diff --git a/mcp/omniread/modules/omniread.json b/mcp/omniread/modules/omniread.json index 9c17528..69dbd9a 100644 --- a/mcp/omniread/modules/omniread.json +++ b/mcp/omniread/modules/omniread.json @@ -8,35 +8,35 @@ "name": "Content", "kind": "class", "path": "omniread.Content", - "signature": "", + "signature": "Content(raw: bytes, source: str, content_type: ContentType | None = ..., metadata: Mapping[str, Any] | None = ...)", "docstring": "Normalized representation of extracted content.\n\nNotes:\n **Responsibilities:**\n\n - A `Content` instance represents a raw content payload along with\n minimal contextual metadata describing its origin and type.\n - This class is the primary exchange format between scrapers,\n parsers, and downstream consumers.", "members": { "raw": { "name": "raw", "kind": "attribute", "path": "omniread.Content.raw", - "signature": "", + "signature": null, "docstring": "Raw content bytes as retrieved from the source." }, "source": { "name": "source", "kind": "attribute", "path": "omniread.Content.source", - "signature": "", + "signature": null, "docstring": "Identifier of the content origin (URL, file path, or logical name)." }, "content_type": { "name": "content_type", "kind": "attribute", "path": "omniread.Content.content_type", - "signature": "", + "signature": null, "docstring": "Optional MIME type of the content, if known." }, "metadata": { "name": "metadata", "kind": "attribute", "path": "omniread.Content.metadata", - "signature": "", + "signature": null, "docstring": "Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes)." } } @@ -45,49 +45,49 @@ "name": "ContentType", "kind": "class", "path": "omniread.ContentType", - "signature": "", + "signature": null, "docstring": "Supported MIME types for extracted content.\n\nNotes:\n **Guarantees:**\n\n - This enum represents the declared or inferred media type of the\n content source.\n - It is primarily used for routing content to the appropriate\n parser or downstream consumer.", "members": { "HTML": { "name": "HTML", "kind": "attribute", "path": "omniread.ContentType.HTML", - "signature": "", + "signature": null, "docstring": "HTML document content." }, "PDF": { "name": "PDF", "kind": "attribute", "path": "omniread.ContentType.PDF", - "signature": "", + "signature": null, "docstring": "PDF document content." }, "XLSX": { "name": "XLSX", "kind": "attribute", "path": "omniread.ContentType.XLSX", - "signature": "", + "signature": null, "docstring": "Office Open XML spreadsheet (xlsx/xlsm) content." }, "CSV": { "name": "CSV", "kind": "attribute", "path": "omniread.ContentType.CSV", - "signature": "", + "signature": null, "docstring": "Comma-separated-value document content." }, "JSON": { "name": "JSON", "kind": "attribute", "path": "omniread.ContentType.JSON", - "signature": "", + "signature": null, "docstring": "JSON document content." }, "XML": { "name": "XML", "kind": "attribute", "path": "omniread.ContentType.XML", - "signature": "", + "signature": null, "docstring": "XML document content." } } @@ -96,14 +96,14 @@ "name": "BaseCsvClient", "kind": "class", "path": "omniread.BaseCsvClient", - "signature": "", + "signature": null, "docstring": "Abstract client responsible for retrieving csv bytes.\n\nRetrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).\n\nNotes:\n **Responsibilities:**\n\n - Implementations must accept a source identifier appropriate to\n the backing store.\n - Return the full csv binary payload.\n - Raise retrieval-specific errors on failure.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.BaseCsvClient.fetch", - "signature": "", + "signature": "fetch(source: Any)", "docstring": "Fetch raw csv bytes from the given source.\n\nArgs:\n source (Any):\n Identifier of the csv location, such as a file path,\n object storage key, or remote reference.\n\nReturns:\n bytes:\n Raw csv bytes.\n\nRaises:\n Exception:\n Retrieval-specific errors defined by the implementation." } } @@ -112,21 +112,21 @@ "name": "CsvParser", "kind": "class", "path": "omniread.CsvParser", - "signature": "", + "signature": "CsvParser(content: Content)", "docstring": "Generic csv parser producing string rows from the document.\n\nNotes:\n **Responsibilities:**\n\n - Decode the payload (UTF-8 with BOM support, Latin-1 fallback).\n - Detect the delimiter from a leading sample (`,` `;` tab `|`),\n defaulting to `,`.\n - Normalize cells into deterministic stripped string values.\n - Expose row extraction helpers mirroring `XlsxParser.rows`.\n\n **Constraints:**\n\n - All values are strings; consumers requiring typed values must\n convert on their side.\n - Quoted fields containing delimiters/newlines are handled by\n the standard ``csv`` module.", "members": { "parse": { "name": "parse", "kind": "function", "path": "omniread.CsvParser.parse", - "signature": "", + "signature": "parse()", "docstring": "Parse the document into normalized string rows.\n\nReturns:\n list[list[str]]:\n Rows of the document." }, "rows": { "name": "rows", "kind": "function", "path": "omniread.CsvParser.rows", - "signature": "", + "signature": "rows(*, skip_empty: bool = True)", "docstring": "Extract normalized string rows from the document.\n\nArgs:\n skip_empty (bool):\n When True (default), rows whose cells are all blank are\n omitted.\n\nReturns:\n list[list[str]]:\n Normalized rows; trailing blank cells are trimmed per row." } } @@ -135,21 +135,21 @@ "name": "CsvParserBase", "kind": "class", "path": "omniread.CsvParserBase", - "signature": "", + "signature": null, "docstring": "Base csv parser.\n\nNotes:\n **Responsibilities:**\n\n - This class enforces csv content-type compatibility and provides\n the extension point for implementing concrete csv parsing\n strategies.\n\n **Constraints:**\n\n - Concrete implementations must define the output type `T` and\n implement the `parse()` method.", "members": { "supported_types": { "name": "supported_types", "kind": "attribute", "path": "omniread.CsvParserBase.supported_types", - "signature": "", + "signature": null, "docstring": "Set of content types supported by this parser (CSV only)." }, "parse": { "name": "parse", "kind": "function", "path": "omniread.CsvParserBase.parse", - "signature": "", + "signature": "parse()", "docstring": "Parse csv content into a structured output.\n\nReturns:\n T:\n Parsed representation of type `T`.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation." } } @@ -158,14 +158,14 @@ "name": "CsvScraper", "kind": "class", "path": "omniread.CsvScraper", - "signature": "", + "signature": "CsvScraper(*, client: BaseCsvClient)", "docstring": "Scraper for csv documents.\n\nNotes:\n **Responsibilities:**\n\n - Fetch raw csv bytes via the configured client.\n - Wrap the payload in a canonical `Content` instance with the\n CSV content type and source identifier.\n\n **Constraints:**\n\n - The scraper does not perform parsing or interpretation.\n - Does not assume a specific storage backend.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.CsvScraper.fetch", - "signature": "", + "signature": "fetch(source: Any, *, metadata: Mapping[str, Any] | None = None)", "docstring": "Fetch a csv document from the given source.\n\nArgs:\n source (Any):\n Identifier of the csv source as understood by the\n configured client.\n metadata (Mapping[str, Any] | None, optional):\n Optional metadata to attach to the returned content.\n\nReturns:\n Content:\n A `Content` instance containing raw csv bytes, source\n identifier, CSV content type, and optional metadata.\n\nRaises:\n Exception:\n Retrieval-specific errors raised by the client." } } @@ -174,14 +174,14 @@ "name": "FileSystemCsvClient", "kind": "class", "path": "omniread.FileSystemCsvClient", - "signature": "", + "signature": null, "docstring": "CSV client that reads from the local filesystem.\n\nNotes:\n **Guarantees:**\n\n - This client reads csv files directly from the disk and\n returns their raw binary contents.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.FileSystemCsvClient.fetch", - "signature": "", + "signature": "fetch(path: Path)", "docstring": "Read a csv file from the local filesystem.\n\nArgs:\n path (Path):\n Filesystem path to the csv file.\n\nReturns:\n bytes:\n Raw csv bytes.\n\nRaises:\n FileNotFoundError:\n If the path does not exist.\n ValueError:\n If the path exists but is not a file." } } @@ -190,28 +190,28 @@ "name": "HTMLScraper", "kind": "class", "path": "omniread.HTMLScraper", - "signature": "", + "signature": "HTMLScraper(*, client: httpx.Client | None = None, timeout: float = 15.0, headers: Mapping[str, str] | None = None, follow_redirects: bool = True)", "docstring": "Base HTML scraper using `httpx`.\n\nNotes:\n **Responsibilities:**\n\n - This scraper retrieves HTML documents over HTTP(S) and returns\n them as raw content wrapped in a `Content` object.\n - Fetches raw bytes and metadata only.\n - The scraper uses `httpx.Client` for HTTP requests, enforces an\n HTML content type, and preserves HTTP response metadata.\n\n **Constraints:**\n\n - The scraper does not: Parse HTML, perform retries or backoff,\n handle non-HTML responses.", "members": { "content_type": { "name": "content_type", "kind": "attribute", "path": "omniread.HTMLScraper.content_type", - "signature": "", + "signature": null, "docstring": null }, "validate_content_type": { "name": "validate_content_type", "kind": "function", "path": "omniread.HTMLScraper.validate_content_type", - "signature": "", + "signature": "validate_content_type(response: httpx.Response)", "docstring": "Validate that the HTTP response contains HTML content.\n\nArgs:\n response (httpx.Response):\n HTTP response returned by `httpx`.\n\nRaises:\n ValueError:\n If the `Content-Type` header is missing or does not indicate HTML content." }, "fetch": { "name": "fetch", "kind": "function", "path": "omniread.HTMLScraper.fetch", - "signature": "", + "signature": "fetch(source: str, *, metadata: Mapping[str, Any] | None = None)", "docstring": "Fetch an HTML document from the given source.\n\nArgs:\n source (str):\n URL of the HTML document.\n metadata (Mapping[str, Any] | None, optional):\n Optional metadata to be merged into the returned content.\n\nReturns:\n Content:\n A `Content` instance containing raw HTML bytes, source URL, HTML content type, and HTTP response metadata.\n\nRaises:\n httpx.HTTPError:\n If the HTTP request fails.\n ValueError:\n If the response is not valid HTML." } } @@ -220,49 +220,49 @@ "name": "HTMLParser", "kind": "class", "path": "omniread.HTMLParser", - "signature": "", + "signature": "HTMLParser(content: Content, features: str = 'html.parser')", "docstring": "Base HTML parser.\n\nNotes:\n **Responsibilities:**\n\n - This class extends the core `BaseParser` with HTML-specific behavior,\n including DOM parsing via BeautifulSoup and reusable extraction helpers.\n - Provides reusable helpers for HTML extraction. Concrete parsers must\n explicitly define the return type.\n\n **Guarantees:**\n\n - Accepts only HTML content.\n - Owns a parsed BeautifulSoup DOM tree.\n - Provides pure helper utilities for common HTML structures.\n\n **Constraints:**\n\n - Concrete subclasses must define the output type `T` and implement\n the `parse()` method.", "members": { "supported_types": { "name": "supported_types", "kind": "attribute", "path": "omniread.HTMLParser.supported_types", - "signature": "", + "signature": null, "docstring": "Set of content types supported by this parser (HTML only)." }, "parse": { "name": "parse", "kind": "function", "path": "omniread.HTMLParser.parse", - "signature": "", + "signature": "parse()", "docstring": "Fully parse the HTML content into structured output.\n\nReturns:\n T:\n Parsed representation of type `T`.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must fully interpret the HTML DOM and return a\n deterministic, structured output." }, "parse_div": { "name": "parse_div", "kind": "function", "path": "omniread.HTMLParser.parse_div", - "signature": "", + "signature": "parse_div(div: Tag, *, separator: str = ' ')", "docstring": "Extract normalized text from a `
` element.\n\nArgs:\n div (Tag):\n BeautifulSoup tag representing a `
`.\n\nReturns:\n list[list[str]]:\n A list of rows, where each row is a list of cell text values." }, "parse_meta": { "name": "parse_meta", "kind": "function", "path": "omniread.HTMLParser.parse_meta", - "signature": "", + "signature": "parse_meta()", "docstring": "Extract high-level metadata from the HTML document.\n\nReturns:\n dict[str, Any]:\n Dictionary containing extracted metadata.\n\nNotes:\n **Responsibilities:**\n\n - Extract high-level metadata from the HTML document.\n - This includes: Document title, `` tag name/property to\n content mappings." } } @@ -271,14 +271,14 @@ "name": "FileSystemPDFClient", "kind": "class", "path": "omniread.FileSystemPDFClient", - "signature": "", + "signature": null, "docstring": "PDF client that reads from the local filesystem.\n\nNotes:\n **Guarantees:**\n\n - This client reads PDF files directly from the disk and returns\n their raw binary contents.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.FileSystemPDFClient.fetch", - "signature": "", + "signature": "fetch(path: Path)", "docstring": "Read a PDF file from the local filesystem.\n\nArgs:\n path (Path):\n Filesystem path to the PDF file.\n\nReturns:\n bytes:\n Raw PDF bytes.\n\nRaises:\n FileNotFoundError:\n If the path does not exist.\n ValueError:\n If the path exists but is not a file." } } @@ -287,14 +287,14 @@ "name": "PDFScraper", "kind": "class", "path": "omniread.PDFScraper", - "signature": "", + "signature": "PDFScraper(*, client: BasePDFClient)", "docstring": "Scraper for PDF sources.\n\nNotes:\n **Responsibilities:**\n\n - Delegates byte retrieval to a PDF client and normalizes output\n into `Content`.\n - Preserves caller-provided metadata.\n\n **Constraints:**\n\n - The scraper does not perform parsing or interpretation.\n - Does not assume a specific storage backend.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.PDFScraper.fetch", - "signature": "", + "signature": "fetch(source: Any, *, metadata: Mapping[str, Any] | None = None)", "docstring": "Fetch a PDF document from the given source.\n\nArgs:\n source (Any):\n Identifier of the PDF source as understood by the configured PDF client.\n metadata (Mapping[str, Any] | None, optional):\n Optional metadata to attach to the returned content.\n\nReturns:\n Content:\n A `Content` instance containing raw PDF bytes, source identifier, PDF content type, and optional metadata.\n\nRaises:\n Exception:\n Retrieval-specific errors raised by the PDF client." } } @@ -303,21 +303,21 @@ "name": "PDFParser", "kind": "class", "path": "omniread.PDFParser", - "signature": "", + "signature": null, "docstring": "Base PDF parser.\n\nNotes:\n **Responsibilities:**\n\n - This class enforces PDF content-type compatibility and provides\n the extension point for implementing concrete PDF parsing strategies.\n\n **Constraints:**\n\n - Concrete implementations must define the output type `T` and\n implement the `parse()` method.", "members": { "supported_types": { "name": "supported_types", "kind": "attribute", "path": "omniread.PDFParser.supported_types", - "signature": "", + "signature": null, "docstring": "Set of content types supported by this parser (PDF only)." }, "parse": { "name": "parse", "kind": "function", "path": "omniread.PDFParser.parse", - "signature": "", + "signature": "parse()", "docstring": "Parse PDF content into a structured output.\n\nReturns:\n T:\n Parsed representation of type `T`.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must fully interpret the PDF binary payload and\n return a deterministic, structured output." } } @@ -326,14 +326,14 @@ "name": "BaseXlsxClient", "kind": "class", "path": "omniread.BaseXlsxClient", - "signature": "", + "signature": null, "docstring": "Abstract client responsible for retrieving spreadsheet bytes.\n\nRetrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).\n\nNotes:\n **Responsibilities:**\n\n - Implementations must accept a source identifier appropriate to\n the backing store.\n - Return the full xlsx binary payload.\n - Raise retrieval-specific errors on failure.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.BaseXlsxClient.fetch", - "signature": "", + "signature": "fetch(source: Any)", "docstring": "Fetch raw xlsx bytes from the given source.\n\nArgs:\n source (Any):\n Identifier of the spreadsheet location, such as a file path,\n object storage key, or remote reference.\n\nReturns:\n bytes:\n Raw xlsx bytes.\n\nRaises:\n Exception:\n Retrieval-specific errors defined by the implementation." } } @@ -342,14 +342,14 @@ "name": "FileSystemXlsxClient", "kind": "class", "path": "omniread.FileSystemXlsxClient", - "signature": "", + "signature": null, "docstring": "XLSX client that reads from the local filesystem.\n\nNotes:\n **Guarantees:**\n\n - This client reads spreadsheet files directly from the disk and\n returns their raw binary contents.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.FileSystemXlsxClient.fetch", - "signature": "", + "signature": "fetch(path: Path)", "docstring": "Read an xlsx file from the local filesystem.\n\nArgs:\n path (Path):\n Filesystem path to the spreadsheet file.\n\nReturns:\n bytes:\n Raw xlsx bytes.\n\nRaises:\n FileNotFoundError:\n If the path does not exist.\n ValueError:\n If the path exists but is not a file." } } @@ -358,35 +358,35 @@ "name": "XlsxParser", "kind": "class", "path": "omniread.XlsxParser", - "signature": "", + "signature": "XlsxParser(content: Content, *, data_only: bool = True, read_only: bool = True)", "docstring": "Generic xlsx parser producing string rows from a worksheet.\n\nNotes:\n **Responsibilities:**\n\n - Lazily load the workbook owned by the parser's content.\n - Normalize cells (including dates and numeric values) into\n deterministic string representations.\n - Expose sheet discovery and row extraction helpers.\n\n **Constraints:**\n\n - Cells are rendered with ``str(value)`` after trimming; date and\n datetime values are rendered in ISO format. Consumers requiring\n locale-specific formatting must convert on their side.", "members": { "workbook": { "name": "workbook", "kind": "attribute", "path": "omniread.XlsxParser.workbook", - "signature": "", + "signature": null, "docstring": "The lazily loaded workbook backing this parser's content." }, "sheet_names": { "name": "sheet_names", "kind": "attribute", "path": "omniread.XlsxParser.sheet_names", - "signature": "", + "signature": null, "docstring": "Names of all worksheets contained in the workbook." }, "parse": { "name": "parse", "kind": "function", "path": "omniread.XlsxParser.parse", - "signature": "", + "signature": "parse()", "docstring": "Parse the first worksheet into normalized string rows.\n\nReturns:\n list[list[str]]:\n Rows of the default (first) worksheet." }, "rows": { "name": "rows", "kind": "function", "path": "omniread.XlsxParser.rows", - "signature": "", + "signature": "rows(sheet: int | str | None = None, *, skip_empty: bool = True)", "docstring": "Extract normalized string rows from a worksheet.\n\nArgs:\n sheet (int | str | None):\n Worksheet index or title; defaults to the first worksheet.\n skip_empty (bool):\n When True (default), rows whose cells are all blank are omitted.\n\nReturns:\n list[list[str]]:\n Normalized rows; trailing blank cells are trimmed per row.\n\nRaises:\n ValueError:\n If the requested sheet does not exist." } } @@ -395,21 +395,21 @@ "name": "XlsxParserBase", "kind": "class", "path": "omniread.XlsxParserBase", - "signature": "", + "signature": null, "docstring": "Base xlsx parser.\n\nNotes:\n **Responsibilities:**\n\n - This class enforces xlsx content-type compatibility and provides\n the extension point for implementing concrete xlsx parsing\n strategies.\n\n **Constraints:**\n\n - Concrete implementations must define the output type `T` and\n implement the `parse()` method.", "members": { "supported_types": { "name": "supported_types", "kind": "attribute", "path": "omniread.XlsxParserBase.supported_types", - "signature": "", + "signature": null, "docstring": "Set of content types supported by this parser (XLSX only)." }, "parse": { "name": "parse", "kind": "function", "path": "omniread.XlsxParserBase.parse", - "signature": "", + "signature": "parse()", "docstring": "Parse xlsx content into a structured output.\n\nReturns:\n T:\n Parsed representation of type `T`.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation." } } @@ -418,14 +418,14 @@ "name": "XlsxScraper", "kind": "class", "path": "omniread.XlsxScraper", - "signature": "", + "signature": "XlsxScraper(*, client: BaseXlsxClient)", "docstring": "Scraper for xlsx spreadsheet documents.\n\nNotes:\n **Responsibilities:**\n\n - Fetch raw xlsx bytes via the configured client.\n - Wrap the payload in a canonical `Content` instance with the\n XLSX content type and source identifier.\n\n **Constraints:**\n\n - The scraper does not perform parsing or interpretation.\n - Does not assume a specific storage backend.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.XlsxScraper.fetch", - "signature": "", + "signature": "fetch(source: Any, *, metadata: Mapping[str, Any] | None = None)", "docstring": "Fetch an xlsx document from the given source.\n\nArgs:\n source (Any):\n Identifier of the spreadsheet source as understood by the\n configured client.\n metadata (Mapping[str, Any] | None, optional):\n Optional metadata to attach to the returned content.\n\nReturns:\n Content:\n A `Content` instance containing raw xlsx bytes, source\n identifier, XLSX content type, and optional metadata.\n\nRaises:\n Exception:\n Retrieval-specific errors raised by the client." } } @@ -441,35 +441,35 @@ "name": "Content", "kind": "class", "path": "omniread.core.Content", - "signature": "", + "signature": "Content(raw: bytes, source: str, content_type: ContentType | None = ..., metadata: Mapping[str, Any] | None = ...)", "docstring": "Normalized representation of extracted content.\n\nNotes:\n **Responsibilities:**\n\n - A `Content` instance represents a raw content payload along with\n minimal contextual metadata describing its origin and type.\n - This class is the primary exchange format between scrapers,\n parsers, and downstream consumers.", "members": { "raw": { "name": "raw", "kind": "attribute", "path": "omniread.core.Content.raw", - "signature": "", + "signature": null, "docstring": "Raw content bytes as retrieved from the source." }, "source": { "name": "source", "kind": "attribute", "path": "omniread.core.Content.source", - "signature": "", + "signature": null, "docstring": "Identifier of the content origin (URL, file path, or logical name)." }, "content_type": { "name": "content_type", "kind": "attribute", "path": "omniread.core.Content.content_type", - "signature": "", + "signature": null, "docstring": "Optional MIME type of the content, if known." }, "metadata": { "name": "metadata", "kind": "attribute", "path": "omniread.core.Content.metadata", - "signature": "", + "signature": null, "docstring": "Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes)." } } @@ -478,49 +478,49 @@ "name": "ContentType", "kind": "class", "path": "omniread.core.ContentType", - "signature": "", + "signature": null, "docstring": "Supported MIME types for extracted content.\n\nNotes:\n **Guarantees:**\n\n - This enum represents the declared or inferred media type of the\n content source.\n - It is primarily used for routing content to the appropriate\n parser or downstream consumer.", "members": { "HTML": { "name": "HTML", "kind": "attribute", "path": "omniread.core.ContentType.HTML", - "signature": "", + "signature": null, "docstring": "HTML document content." }, "PDF": { "name": "PDF", "kind": "attribute", "path": "omniread.core.ContentType.PDF", - "signature": "", + "signature": null, "docstring": "PDF document content." }, "XLSX": { "name": "XLSX", "kind": "attribute", "path": "omniread.core.ContentType.XLSX", - "signature": "", + "signature": null, "docstring": "Office Open XML spreadsheet (xlsx/xlsm) content." }, "CSV": { "name": "CSV", "kind": "attribute", "path": "omniread.core.ContentType.CSV", - "signature": "", + "signature": null, "docstring": "Comma-separated-value document content." }, "JSON": { "name": "JSON", "kind": "attribute", "path": "omniread.core.ContentType.JSON", - "signature": "", + "signature": null, "docstring": "JSON document content." }, "XML": { "name": "XML", "kind": "attribute", "path": "omniread.core.ContentType.XML", - "signature": "", + "signature": null, "docstring": "XML document content." } } @@ -529,35 +529,35 @@ "name": "BaseParser", "kind": "class", "path": "omniread.core.BaseParser", - "signature": "", + "signature": "BaseParser(content: Content)", "docstring": "Base interface for all parsers.\n\nNotes:\n **Guarantees:**\n\n - A parser is a self-contained object that owns the `Content` it is\n responsible for interpreting.\n - Consumers may rely on early validation of content compatibility\n and type-stable return values from `parse()`.\n\n **Responsibilities:**\n\n - Implementations must declare supported content types via `supported_types`.\n - Implementations must raise parsing-specific exceptions from `parse()`.\n - Implementations must remain deterministic for a given input.", "members": { "supported_types": { "name": "supported_types", "kind": "attribute", "path": "omniread.core.BaseParser.supported_types", - "signature": "", + "signature": null, "docstring": "Set of content types supported by this parser. An empty set indicates that the parser is content-type agnostic." }, "content": { "name": "content", "kind": "attribute", "path": "omniread.core.BaseParser.content", - "signature": "", + "signature": null, "docstring": null }, "parse": { "name": "parse", "kind": "function", "path": "omniread.core.BaseParser.parse", - "signature": "", + "signature": "parse()", "docstring": "Parse the owned content into structured output.\n\nReturns:\n T:\n Parsed, structured representation.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must fully consume the provided content and\n return a deterministic, structured output." }, "supports": { "name": "supports", "kind": "function", "path": "omniread.core.BaseParser.supports", - "signature": "", + "signature": "supports()", "docstring": "Check whether this parser supports the content's type.\n\nReturns:\n bool:\n True if the content type is supported; False otherwise." } } @@ -566,14 +566,14 @@ "name": "BaseScraper", "kind": "class", "path": "omniread.core.BaseScraper", - "signature": "", + "signature": null, "docstring": "Base interface for all scrapers.\n\nNotes:\n **Responsibilities:**\n\n - A scraper is responsible ONLY for fetching raw content (bytes)\n from a source. It must not interpret or parse it.\n - A scraper is a stateless acquisition component that retrieves raw\n content from a source and returns it as a `Content` object.\n - Scrapers define how content is obtained, not what the content means.\n - Implementations may vary in transport mechanism, authentication\n strategy, retry and backoff behavior.\n\n **Constraints:**\n\n - Implementations must not parse content, modify content semantics,\n or couple scraping logic to a specific parser.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.core.BaseScraper.fetch", - "signature": "", + "signature": "fetch(source: str, *, metadata: Mapping[str, Any] | None = None)", "docstring": "Fetch raw content from the given source.\n\nArgs:\n source (str):\n Location identifier (URL, file path, S3 URI, etc.).\n\n metadata (Mapping[str, Any] | None, optional):\n Optional hints for the scraper (headers, auth, etc.).\n\nReturns:\n Content:\n Content object containing raw bytes and metadata.\n\nRaises:\n Exception:\n Retrieval-specific errors as defined by the implementation.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must retrieve the content referenced by `source`\n and return it as raw bytes wrapped in a `Content` object." } } @@ -585,39 +585,11 @@ "signature": null, "docstring": "# Summary\n\nCanonical content models for OmniRead.\n\nThis module defines the **format-agnostic content representation** used across\nall parsers and scrapers in OmniRead.\n\nThe models defined here represent *what* was extracted, not *how* it was\nretrieved or parsed. Format-specific behavior and metadata must not alter\nthe semantic meaning of these models.", "members": { - "Mapping": { - "name": "Mapping", - "kind": "alias", - "path": "omniread.core.content.Mapping", - "signature": "", - "docstring": null - }, - "dataclass": { - "name": "dataclass", - "kind": "alias", - "path": "omniread.core.content.dataclass", - "signature": "", - "docstring": null - }, - "Enum": { - "name": "Enum", - "kind": "alias", - "path": "omniread.core.content.Enum", - "signature": "", - "docstring": null - }, - "Any": { - "name": "Any", - "kind": "alias", - "path": "omniread.core.content.Any", - "signature": "", - "docstring": null - }, "ContentType": { "name": "ContentType", "kind": "class", "path": "omniread.core.content.ContentType", - "signature": "", + "signature": null, "docstring": "Supported MIME types for extracted content.\n\nNotes:\n **Guarantees:**\n\n - This enum represents the declared or inferred media type of the\n content source.\n - It is primarily used for routing content to the appropriate\n parser or downstream consumer.", "members": { "HTML": { @@ -668,7 +640,7 @@ "name": "Content", "kind": "class", "path": "omniread.core.content.Content", - "signature": "", + "signature": "Content(raw: bytes, source: str, content_type: ContentType | None = ..., metadata: Mapping[str, Any] | None = ...)", "docstring": "Normalized representation of extracted content.\n\nNotes:\n **Responsibilities:**\n\n - A `Content` instance represents a raw content payload along with\n minimal contextual metadata describing its origin and type.\n - This class is the primary exchange format between scrapers,\n parsers, and downstream consumers.", "members": { "raw": { @@ -710,67 +682,39 @@ "signature": null, "docstring": "# Summary\n\nAbstract parsing contracts for OmniRead.\n\nThis module defines the **format-agnostic parser interface** used to transform\nraw content into structured, typed representations.\n\nParsers are responsible for:\n\n- Interpreting a single `Content` instance\n- Validating compatibility with the content type\n- Producing a structured output suitable for downstream consumers\n\nParsers are not responsible for:\n\n- Fetching or acquiring content\n- Performing retries or error recovery\n- Managing multiple content sources", "members": { - "ABC": { - "name": "ABC", - "kind": "alias", - "path": "omniread.core.parser.ABC", - "signature": "", - "docstring": null - }, - "abstractmethod": { - "name": "abstractmethod", - "kind": "alias", - "path": "omniread.core.parser.abstractmethod", - "signature": "", - "docstring": null - }, - "Generic": { - "name": "Generic", - "kind": "alias", - "path": "omniread.core.parser.Generic", - "signature": "", - "docstring": null - }, - "TypeVar": { - "name": "TypeVar", - "kind": "alias", - "path": "omniread.core.parser.TypeVar", - "signature": "", - "docstring": null - }, "Content": { "name": "Content", "kind": "class", "path": "omniread.core.parser.Content", - "signature": "", + "signature": "Content(raw: bytes, source: str, content_type: ContentType | None = ..., metadata: Mapping[str, Any] | None = ...)", "docstring": "Normalized representation of extracted content.\n\nNotes:\n **Responsibilities:**\n\n - A `Content` instance represents a raw content payload along with\n minimal contextual metadata describing its origin and type.\n - This class is the primary exchange format between scrapers,\n parsers, and downstream consumers.", "members": { "raw": { "name": "raw", "kind": "attribute", "path": "omniread.core.parser.Content.raw", - "signature": "", + "signature": null, "docstring": "Raw content bytes as retrieved from the source." }, "source": { "name": "source", "kind": "attribute", "path": "omniread.core.parser.Content.source", - "signature": "", + "signature": null, "docstring": "Identifier of the content origin (URL, file path, or logical name)." }, "content_type": { "name": "content_type", "kind": "attribute", "path": "omniread.core.parser.Content.content_type", - "signature": "", + "signature": null, "docstring": "Optional MIME type of the content, if known." }, "metadata": { "name": "metadata", "kind": "attribute", "path": "omniread.core.parser.Content.metadata", - "signature": "", + "signature": null, "docstring": "Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes)." } } @@ -779,49 +723,49 @@ "name": "ContentType", "kind": "class", "path": "omniread.core.parser.ContentType", - "signature": "", + "signature": null, "docstring": "Supported MIME types for extracted content.\n\nNotes:\n **Guarantees:**\n\n - This enum represents the declared or inferred media type of the\n content source.\n - It is primarily used for routing content to the appropriate\n parser or downstream consumer.", "members": { "HTML": { "name": "HTML", "kind": "attribute", "path": "omniread.core.parser.ContentType.HTML", - "signature": "", + "signature": null, "docstring": "HTML document content." }, "PDF": { "name": "PDF", "kind": "attribute", "path": "omniread.core.parser.ContentType.PDF", - "signature": "", + "signature": null, "docstring": "PDF document content." }, "XLSX": { "name": "XLSX", "kind": "attribute", "path": "omniread.core.parser.ContentType.XLSX", - "signature": "", + "signature": null, "docstring": "Office Open XML spreadsheet (xlsx/xlsm) content." }, "CSV": { "name": "CSV", "kind": "attribute", "path": "omniread.core.parser.ContentType.CSV", - "signature": "", + "signature": null, "docstring": "Comma-separated-value document content." }, "JSON": { "name": "JSON", "kind": "attribute", "path": "omniread.core.parser.ContentType.JSON", - "signature": "", + "signature": null, "docstring": "JSON document content." }, "XML": { "name": "XML", "kind": "attribute", "path": "omniread.core.parser.ContentType.XML", - "signature": "", + "signature": null, "docstring": "XML document content." } } @@ -837,7 +781,7 @@ "name": "BaseParser", "kind": "class", "path": "omniread.core.parser.BaseParser", - "signature": "", + "signature": "BaseParser(content: Content)", "docstring": "Base interface for all parsers.\n\nNotes:\n **Guarantees:**\n\n - A parser is a self-contained object that owns the `Content` it is\n responsible for interpreting.\n - Consumers may rely on early validation of content compatibility\n and type-stable return values from `parse()`.\n\n **Responsibilities:**\n\n - Implementations must declare supported content types via `supported_types`.\n - Implementations must raise parsing-specific exceptions from `parse()`.\n - Implementations must remain deterministic for a given input.", "members": { "supported_types": { @@ -858,14 +802,14 @@ "name": "parse", "kind": "function", "path": "omniread.core.parser.BaseParser.parse", - "signature": "", + "signature": "parse() -> T", "docstring": "Parse the owned content into structured output.\n\nReturns:\n T:\n Parsed, structured representation.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must fully consume the provided content and\n return a deterministic, structured output." }, "supports": { "name": "supports", "kind": "function", "path": "omniread.core.parser.BaseParser.supports", - "signature": "", + "signature": "supports() -> bool", "docstring": "Check whether this parser supports the content's type.\n\nReturns:\n bool:\n True if the content type is supported; False otherwise." } } @@ -879,67 +823,39 @@ "signature": null, "docstring": "# Summary\n\nAbstract scraping contracts for OmniRead.\n\nThis module defines the **format-agnostic scraper interface** responsible for\nacquiring raw content from external sources.\n\nScrapers are responsible for:\n\n- Locating and retrieving raw content bytes\n- Attaching minimal contextual metadata\n- Returning normalized `Content` objects\n\nScrapers are explicitly NOT responsible for:\n\n- Parsing or interpreting content\n- Inferring structure or semantics\n- Performing content-type specific processing\n\nAll interpretation must be delegated to parsers.", "members": { - "ABC": { - "name": "ABC", - "kind": "alias", - "path": "omniread.core.scraper.ABC", - "signature": "", - "docstring": null - }, - "abstractmethod": { - "name": "abstractmethod", - "kind": "alias", - "path": "omniread.core.scraper.abstractmethod", - "signature": "", - "docstring": null - }, - "Mapping": { - "name": "Mapping", - "kind": "alias", - "path": "omniread.core.scraper.Mapping", - "signature": "", - "docstring": null - }, - "Any": { - "name": "Any", - "kind": "alias", - "path": "omniread.core.scraper.Any", - "signature": "", - "docstring": null - }, "Content": { "name": "Content", "kind": "class", "path": "omniread.core.scraper.Content", - "signature": "", + "signature": "Content(raw: bytes, source: str, content_type: ContentType | None = ..., metadata: Mapping[str, Any] | None = ...)", "docstring": "Normalized representation of extracted content.\n\nNotes:\n **Responsibilities:**\n\n - A `Content` instance represents a raw content payload along with\n minimal contextual metadata describing its origin and type.\n - This class is the primary exchange format between scrapers,\n parsers, and downstream consumers.", "members": { "raw": { "name": "raw", "kind": "attribute", "path": "omniread.core.scraper.Content.raw", - "signature": "", + "signature": null, "docstring": "Raw content bytes as retrieved from the source." }, "source": { "name": "source", "kind": "attribute", "path": "omniread.core.scraper.Content.source", - "signature": "", + "signature": null, "docstring": "Identifier of the content origin (URL, file path, or logical name)." }, "content_type": { "name": "content_type", "kind": "attribute", "path": "omniread.core.scraper.Content.content_type", - "signature": "", + "signature": null, "docstring": "Optional MIME type of the content, if known." }, "metadata": { "name": "metadata", "kind": "attribute", "path": "omniread.core.scraper.Content.metadata", - "signature": "", + "signature": null, "docstring": "Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes)." } } @@ -948,14 +864,14 @@ "name": "BaseScraper", "kind": "class", "path": "omniread.core.scraper.BaseScraper", - "signature": "", + "signature": null, "docstring": "Base interface for all scrapers.\n\nNotes:\n **Responsibilities:**\n\n - A scraper is responsible ONLY for fetching raw content (bytes)\n from a source. It must not interpret or parse it.\n - A scraper is a stateless acquisition component that retrieves raw\n content from a source and returns it as a `Content` object.\n - Scrapers define how content is obtained, not what the content means.\n - Implementations may vary in transport mechanism, authentication\n strategy, retry and backoff behavior.\n\n **Constraints:**\n\n - Implementations must not parse content, modify content semantics,\n or couple scraping logic to a specific parser.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.core.scraper.BaseScraper.fetch", - "signature": "", + "signature": "fetch(source: str, *, metadata: Mapping[str, Any] | None = None) -> Content", "docstring": "Fetch raw content from the given source.\n\nArgs:\n source (str):\n Location identifier (URL, file path, S3 URI, etc.).\n\n metadata (Mapping[str, Any] | None, optional):\n Optional hints for the scraper (headers, auth, etc.).\n\nReturns:\n Content:\n Content object containing raw bytes and metadata.\n\nRaises:\n Exception:\n Retrieval-specific errors as defined by the implementation.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must retrieve the content referenced by `source`\n and return it as raw bytes wrapped in a `Content` object." } } @@ -975,14 +891,14 @@ "name": "BaseCsvClient", "kind": "class", "path": "omniread.csv.BaseCsvClient", - "signature": "", + "signature": null, "docstring": "Abstract client responsible for retrieving csv bytes.\n\nRetrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).\n\nNotes:\n **Responsibilities:**\n\n - Implementations must accept a source identifier appropriate to\n the backing store.\n - Return the full csv binary payload.\n - Raise retrieval-specific errors on failure.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.csv.BaseCsvClient.fetch", - "signature": "", + "signature": "fetch(source: Any)", "docstring": "Fetch raw csv bytes from the given source.\n\nArgs:\n source (Any):\n Identifier of the csv location, such as a file path,\n object storage key, or remote reference.\n\nReturns:\n bytes:\n Raw csv bytes.\n\nRaises:\n Exception:\n Retrieval-specific errors defined by the implementation." } } @@ -991,14 +907,14 @@ "name": "FileSystemCsvClient", "kind": "class", "path": "omniread.csv.FileSystemCsvClient", - "signature": "", + "signature": null, "docstring": "CSV client that reads from the local filesystem.\n\nNotes:\n **Guarantees:**\n\n - This client reads csv files directly from the disk and\n returns their raw binary contents.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.csv.FileSystemCsvClient.fetch", - "signature": "", + "signature": "fetch(path: Path)", "docstring": "Read a csv file from the local filesystem.\n\nArgs:\n path (Path):\n Filesystem path to the csv file.\n\nReturns:\n bytes:\n Raw csv bytes.\n\nRaises:\n FileNotFoundError:\n If the path does not exist.\n ValueError:\n If the path exists but is not a file." } } @@ -1007,21 +923,21 @@ "name": "CsvParser", "kind": "class", "path": "omniread.csv.CsvParser", - "signature": "", + "signature": "CsvParser(content: Content)", "docstring": "Generic csv parser producing string rows from the document.\n\nNotes:\n **Responsibilities:**\n\n - Decode the payload (UTF-8 with BOM support, Latin-1 fallback).\n - Detect the delimiter from a leading sample (`,` `;` tab `|`),\n defaulting to `,`.\n - Normalize cells into deterministic stripped string values.\n - Expose row extraction helpers mirroring `XlsxParser.rows`.\n\n **Constraints:**\n\n - All values are strings; consumers requiring typed values must\n convert on their side.\n - Quoted fields containing delimiters/newlines are handled by\n the standard ``csv`` module.", "members": { "parse": { "name": "parse", "kind": "function", "path": "omniread.csv.CsvParser.parse", - "signature": "", + "signature": "parse()", "docstring": "Parse the document into normalized string rows.\n\nReturns:\n list[list[str]]:\n Rows of the document." }, "rows": { "name": "rows", "kind": "function", "path": "omniread.csv.CsvParser.rows", - "signature": "", + "signature": "rows(*, skip_empty: bool = True)", "docstring": "Extract normalized string rows from the document.\n\nArgs:\n skip_empty (bool):\n When True (default), rows whose cells are all blank are\n omitted.\n\nReturns:\n list[list[str]]:\n Normalized rows; trailing blank cells are trimmed per row." } } @@ -1030,21 +946,21 @@ "name": "CsvParserBase", "kind": "class", "path": "omniread.csv.CsvParserBase", - "signature": "", + "signature": null, "docstring": "Base csv parser.\n\nNotes:\n **Responsibilities:**\n\n - This class enforces csv content-type compatibility and provides\n the extension point for implementing concrete csv parsing\n strategies.\n\n **Constraints:**\n\n - Concrete implementations must define the output type `T` and\n implement the `parse()` method.", "members": { "supported_types": { "name": "supported_types", "kind": "attribute", "path": "omniread.csv.CsvParserBase.supported_types", - "signature": "", + "signature": null, "docstring": "Set of content types supported by this parser (CSV only)." }, "parse": { "name": "parse", "kind": "function", "path": "omniread.csv.CsvParserBase.parse", - "signature": "", + "signature": "parse()", "docstring": "Parse csv content into a structured output.\n\nReturns:\n T:\n Parsed representation of type `T`.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation." } } @@ -1053,14 +969,14 @@ "name": "CsvScraper", "kind": "class", "path": "omniread.csv.CsvScraper", - "signature": "", + "signature": "CsvScraper(*, client: BaseCsvClient)", "docstring": "Scraper for csv documents.\n\nNotes:\n **Responsibilities:**\n\n - Fetch raw csv bytes via the configured client.\n - Wrap the payload in a canonical `Content` instance with the\n CSV content type and source identifier.\n\n **Constraints:**\n\n - The scraper does not perform parsing or interpretation.\n - Does not assume a specific storage backend.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.csv.CsvScraper.fetch", - "signature": "", + "signature": "fetch(source: Any, *, metadata: Mapping[str, Any] | None = None)", "docstring": "Fetch a csv document from the given source.\n\nArgs:\n source (Any):\n Identifier of the csv source as understood by the\n configured client.\n metadata (Mapping[str, Any] | None, optional):\n Optional metadata to attach to the returned content.\n\nReturns:\n Content:\n A `Content` instance containing raw csv bytes, source\n identifier, CSV content type, and optional metadata.\n\nRaises:\n Exception:\n Retrieval-specific errors raised by the client." } } @@ -1072,46 +988,18 @@ "signature": null, "docstring": "# Summary\n\nCSV client abstractions for OmniRead.\n\nThis module defines the **client layer** responsible for retrieving raw\ncomma-separated-value document bytes from a concrete backing store.\n\nClients provide low-level access to csv binaries and are intentionally\ndecoupled from scraping and parsing logic. They do not perform validation,\ninterpretation, or content extraction.\n\nTypical backing stores include:\n\n- Local filesystems\n- Object storage (S3, GCS, etc.)\n- Network file systems", "members": { - "ABC": { - "name": "ABC", - "kind": "alias", - "path": "omniread.csv.client.ABC", - "signature": "", - "docstring": null - }, - "abstractmethod": { - "name": "abstractmethod", - "kind": "alias", - "path": "omniread.csv.client.abstractmethod", - "signature": "", - "docstring": null - }, - "Path": { - "name": "Path", - "kind": "alias", - "path": "omniread.csv.client.Path", - "signature": "", - "docstring": null - }, - "Any": { - "name": "Any", - "kind": "alias", - "path": "omniread.csv.client.Any", - "signature": "", - "docstring": null - }, "BaseCsvClient": { "name": "BaseCsvClient", "kind": "class", "path": "omniread.csv.client.BaseCsvClient", - "signature": "", + "signature": null, "docstring": "Abstract client responsible for retrieving csv bytes.\n\nRetrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).\n\nNotes:\n **Responsibilities:**\n\n - Implementations must accept a source identifier appropriate to\n the backing store.\n - Return the full csv binary payload.\n - Raise retrieval-specific errors on failure.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.csv.client.BaseCsvClient.fetch", - "signature": "", + "signature": "fetch(source: Any) -> bytes", "docstring": "Fetch raw csv bytes from the given source.\n\nArgs:\n source (Any):\n Identifier of the csv location, such as a file path,\n object storage key, or remote reference.\n\nReturns:\n bytes:\n Raw csv bytes.\n\nRaises:\n Exception:\n Retrieval-specific errors defined by the implementation." } } @@ -1120,14 +1008,14 @@ "name": "FileSystemCsvClient", "kind": "class", "path": "omniread.csv.client.FileSystemCsvClient", - "signature": "", + "signature": null, "docstring": "CSV client that reads from the local filesystem.\n\nNotes:\n **Guarantees:**\n\n - This client reads csv files directly from the disk and\n returns their raw binary contents.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.csv.client.FileSystemCsvClient.fetch", - "signature": "", + "signature": "fetch(path: Path) -> bytes", "docstring": "Read a csv file from the local filesystem.\n\nArgs:\n path (Path):\n Filesystem path to the csv file.\n\nReturns:\n bytes:\n Raw csv bytes.\n\nRaises:\n FileNotFoundError:\n If the path does not exist.\n ValueError:\n If the path exists but is not a file." } } @@ -1141,60 +1029,39 @@ "signature": null, "docstring": "# Summary\n\nCSV parser implementations for OmniRead.\n\nThis module provides a concrete, generic parser for comma-separated-value\ndocuments. It exposes records as lists of string cells so downstream\nconsumers can interpret tabular content without depending on the ``csv``\nmodule directly.\n\nThe parser is intentionally statement-agnostic: it performs no header\ndetection or column interpretation beyond basic cell normalization and\ndelimiter detection.", "members": { - "Sniffer": { - "name": "Sniffer", - "kind": "alias", - "path": "omniread.csv.parser.Sniffer", - "signature": "", - "docstring": null - }, - "reader": { - "name": "reader", - "kind": "alias", - "path": "omniread.csv.parser.reader", - "signature": "", - "docstring": null - }, - "StringIO": { - "name": "StringIO", - "kind": "alias", - "path": "omniread.csv.parser.StringIO", - "signature": "", - "docstring": null - }, "Content": { "name": "Content", "kind": "class", "path": "omniread.csv.parser.Content", - "signature": "", + "signature": "Content(raw: bytes, source: str, content_type: ContentType | None = ..., metadata: Mapping[str, Any] | None = ...)", "docstring": "Normalized representation of extracted content.\n\nNotes:\n **Responsibilities:**\n\n - A `Content` instance represents a raw content payload along with\n minimal contextual metadata describing its origin and type.\n - This class is the primary exchange format between scrapers,\n parsers, and downstream consumers.", "members": { "raw": { "name": "raw", "kind": "attribute", "path": "omniread.csv.parser.Content.raw", - "signature": "", + "signature": null, "docstring": "Raw content bytes as retrieved from the source." }, "source": { "name": "source", "kind": "attribute", "path": "omniread.csv.parser.Content.source", - "signature": "", + "signature": null, "docstring": "Identifier of the content origin (URL, file path, or logical name)." }, "content_type": { "name": "content_type", "kind": "attribute", "path": "omniread.csv.parser.Content.content_type", - "signature": "", + "signature": null, "docstring": "Optional MIME type of the content, if known." }, "metadata": { "name": "metadata", "kind": "attribute", "path": "omniread.csv.parser.Content.metadata", - "signature": "", + "signature": null, "docstring": "Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes)." } } @@ -1203,21 +1070,21 @@ "name": "CsvParserBase", "kind": "class", "path": "omniread.csv.parser.CsvParserBase", - "signature": "", + "signature": null, "docstring": "Base csv parser.\n\nNotes:\n **Responsibilities:**\n\n - This class enforces csv content-type compatibility and provides\n the extension point for implementing concrete csv parsing\n strategies.\n\n **Constraints:**\n\n - Concrete implementations must define the output type `T` and\n implement the `parse()` method.", "members": { "supported_types": { "name": "supported_types", "kind": "attribute", "path": "omniread.csv.parser.CsvParserBase.supported_types", - "signature": "", + "signature": null, "docstring": "Set of content types supported by this parser (CSV only)." }, "parse": { "name": "parse", "kind": "function", "path": "omniread.csv.parser.CsvParserBase.parse", - "signature": "", + "signature": "parse()", "docstring": "Parse csv content into a structured output.\n\nReturns:\n T:\n Parsed representation of type `T`.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation." } } @@ -1226,21 +1093,21 @@ "name": "CsvParser", "kind": "class", "path": "omniread.csv.parser.CsvParser", - "signature": "", + "signature": "CsvParser(content: Content)", "docstring": "Generic csv parser producing string rows from the document.\n\nNotes:\n **Responsibilities:**\n\n - Decode the payload (UTF-8 with BOM support, Latin-1 fallback).\n - Detect the delimiter from a leading sample (`,` `;` tab `|`),\n defaulting to `,`.\n - Normalize cells into deterministic stripped string values.\n - Expose row extraction helpers mirroring `XlsxParser.rows`.\n\n **Constraints:**\n\n - All values are strings; consumers requiring typed values must\n convert on their side.\n - Quoted fields containing delimiters/newlines are handled by\n the standard ``csv`` module.", "members": { "parse": { "name": "parse", "kind": "function", "path": "omniread.csv.parser.CsvParser.parse", - "signature": "", + "signature": "parse() -> list[list[str]]", "docstring": "Parse the document into normalized string rows.\n\nReturns:\n list[list[str]]:\n Rows of the document." }, "rows": { "name": "rows", "kind": "function", "path": "omniread.csv.parser.CsvParser.rows", - "signature": "", + "signature": "rows(*, skip_empty: bool = True) -> list[list[str]]", "docstring": "Extract normalized string rows from the document.\n\nArgs:\n skip_empty (bool):\n When True (default), rows whose cells are all blank are\n omitted.\n\nReturns:\n list[list[str]]:\n Normalized rows; trailing blank cells are trimmed per row." } } @@ -1254,74 +1121,53 @@ "signature": null, "docstring": "# Summary\n\nCSV parser base implementation for OmniRead.\n\nThis module defines the **CSV-specific parser contract**, extending the\nformat-agnostic `BaseParser` with constraints appropriate for\ncomma-separated-value documents.", "members": { - "abstractmethod": { - "name": "abstractmethod", - "kind": "alias", - "path": "omniread.csv.parser_base.abstractmethod", - "signature": "", - "docstring": null - }, - "Generic": { - "name": "Generic", - "kind": "alias", - "path": "omniread.csv.parser_base.Generic", - "signature": "", - "docstring": null - }, - "TypeVar": { - "name": "TypeVar", - "kind": "alias", - "path": "omniread.csv.parser_base.TypeVar", - "signature": "", - "docstring": null - }, "ContentType": { "name": "ContentType", "kind": "class", "path": "omniread.csv.parser_base.ContentType", - "signature": "", + "signature": null, "docstring": "Supported MIME types for extracted content.\n\nNotes:\n **Guarantees:**\n\n - This enum represents the declared or inferred media type of the\n content source.\n - It is primarily used for routing content to the appropriate\n parser or downstream consumer.", "members": { "HTML": { "name": "HTML", "kind": "attribute", "path": "omniread.csv.parser_base.ContentType.HTML", - "signature": "", + "signature": null, "docstring": "HTML document content." }, "PDF": { "name": "PDF", "kind": "attribute", "path": "omniread.csv.parser_base.ContentType.PDF", - "signature": "", + "signature": null, "docstring": "PDF document content." }, "XLSX": { "name": "XLSX", "kind": "attribute", "path": "omniread.csv.parser_base.ContentType.XLSX", - "signature": "", + "signature": null, "docstring": "Office Open XML spreadsheet (xlsx/xlsm) content." }, "CSV": { "name": "CSV", "kind": "attribute", "path": "omniread.csv.parser_base.ContentType.CSV", - "signature": "", + "signature": null, "docstring": "Comma-separated-value document content." }, "JSON": { "name": "JSON", "kind": "attribute", "path": "omniread.csv.parser_base.ContentType.JSON", - "signature": "", + "signature": null, "docstring": "JSON document content." }, "XML": { "name": "XML", "kind": "attribute", "path": "omniread.csv.parser_base.ContentType.XML", - "signature": "", + "signature": null, "docstring": "XML document content." } } @@ -1330,35 +1176,35 @@ "name": "BaseParser", "kind": "class", "path": "omniread.csv.parser_base.BaseParser", - "signature": "", + "signature": "BaseParser(content: Content)", "docstring": "Base interface for all parsers.\n\nNotes:\n **Guarantees:**\n\n - A parser is a self-contained object that owns the `Content` it is\n responsible for interpreting.\n - Consumers may rely on early validation of content compatibility\n and type-stable return values from `parse()`.\n\n **Responsibilities:**\n\n - Implementations must declare supported content types via `supported_types`.\n - Implementations must raise parsing-specific exceptions from `parse()`.\n - Implementations must remain deterministic for a given input.", "members": { "supported_types": { "name": "supported_types", "kind": "attribute", "path": "omniread.csv.parser_base.BaseParser.supported_types", - "signature": "", + "signature": null, "docstring": "Set of content types supported by this parser. An empty set indicates that the parser is content-type agnostic." }, "content": { "name": "content", "kind": "attribute", "path": "omniread.csv.parser_base.BaseParser.content", - "signature": "", + "signature": null, "docstring": null }, "parse": { "name": "parse", "kind": "function", "path": "omniread.csv.parser_base.BaseParser.parse", - "signature": "", + "signature": "parse()", "docstring": "Parse the owned content into structured output.\n\nReturns:\n T:\n Parsed, structured representation.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must fully consume the provided content and\n return a deterministic, structured output." }, "supports": { "name": "supports", "kind": "function", "path": "omniread.csv.parser_base.BaseParser.supports", - "signature": "", + "signature": "supports()", "docstring": "Check whether this parser supports the content's type.\n\nReturns:\n bool:\n True if the content type is supported; False otherwise." } } @@ -1374,7 +1220,7 @@ "name": "CsvParserBase", "kind": "class", "path": "omniread.csv.parser_base.CsvParserBase", - "signature": "", + "signature": null, "docstring": "Base csv parser.\n\nNotes:\n **Responsibilities:**\n\n - This class enforces csv content-type compatibility and provides\n the extension point for implementing concrete csv parsing\n strategies.\n\n **Constraints:**\n\n - Concrete implementations must define the output type `T` and\n implement the `parse()` method.", "members": { "supported_types": { @@ -1388,7 +1234,7 @@ "name": "parse", "kind": "function", "path": "omniread.csv.parser_base.CsvParserBase.parse", - "signature": "", + "signature": "parse() -> T", "docstring": "Parse csv content into a structured output.\n\nReturns:\n T:\n Parsed representation of type `T`.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation." } } @@ -1402,53 +1248,39 @@ "signature": null, "docstring": "# Summary\n\nCSV scraper for OmniRead.\n\nThis module defines the scraper responsible for acquiring raw\ncomma-separated-value document content from a backing store via a\nconfigured client.\n\nThe scraper does not interpret or parse the acquired bytes; it wraps them in\nthe canonical `Content` model.", "members": { - "Mapping": { - "name": "Mapping", - "kind": "alias", - "path": "omniread.csv.scraper.Mapping", - "signature": "", - "docstring": null - }, - "Any": { - "name": "Any", - "kind": "alias", - "path": "omniread.csv.scraper.Any", - "signature": "", - "docstring": null - }, "Content": { "name": "Content", "kind": "class", "path": "omniread.csv.scraper.Content", - "signature": "", + "signature": "Content(raw: bytes, source: str, content_type: ContentType | None = ..., metadata: Mapping[str, Any] | None = ...)", "docstring": "Normalized representation of extracted content.\n\nNotes:\n **Responsibilities:**\n\n - A `Content` instance represents a raw content payload along with\n minimal contextual metadata describing its origin and type.\n - This class is the primary exchange format between scrapers,\n parsers, and downstream consumers.", "members": { "raw": { "name": "raw", "kind": "attribute", "path": "omniread.csv.scraper.Content.raw", - "signature": "", + "signature": null, "docstring": "Raw content bytes as retrieved from the source." }, "source": { "name": "source", "kind": "attribute", "path": "omniread.csv.scraper.Content.source", - "signature": "", + "signature": null, "docstring": "Identifier of the content origin (URL, file path, or logical name)." }, "content_type": { "name": "content_type", "kind": "attribute", "path": "omniread.csv.scraper.Content.content_type", - "signature": "", + "signature": null, "docstring": "Optional MIME type of the content, if known." }, "metadata": { "name": "metadata", "kind": "attribute", "path": "omniread.csv.scraper.Content.metadata", - "signature": "", + "signature": null, "docstring": "Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes)." } } @@ -1457,49 +1289,49 @@ "name": "ContentType", "kind": "class", "path": "omniread.csv.scraper.ContentType", - "signature": "", + "signature": null, "docstring": "Supported MIME types for extracted content.\n\nNotes:\n **Guarantees:**\n\n - This enum represents the declared or inferred media type of the\n content source.\n - It is primarily used for routing content to the appropriate\n parser or downstream consumer.", "members": { "HTML": { "name": "HTML", "kind": "attribute", "path": "omniread.csv.scraper.ContentType.HTML", - "signature": "", + "signature": null, "docstring": "HTML document content." }, "PDF": { "name": "PDF", "kind": "attribute", "path": "omniread.csv.scraper.ContentType.PDF", - "signature": "", + "signature": null, "docstring": "PDF document content." }, "XLSX": { "name": "XLSX", "kind": "attribute", "path": "omniread.csv.scraper.ContentType.XLSX", - "signature": "", + "signature": null, "docstring": "Office Open XML spreadsheet (xlsx/xlsm) content." }, "CSV": { "name": "CSV", "kind": "attribute", "path": "omniread.csv.scraper.ContentType.CSV", - "signature": "", + "signature": null, "docstring": "Comma-separated-value document content." }, "JSON": { "name": "JSON", "kind": "attribute", "path": "omniread.csv.scraper.ContentType.JSON", - "signature": "", + "signature": null, "docstring": "JSON document content." }, "XML": { "name": "XML", "kind": "attribute", "path": "omniread.csv.scraper.ContentType.XML", - "signature": "", + "signature": null, "docstring": "XML document content." } } @@ -1508,14 +1340,14 @@ "name": "BaseCsvClient", "kind": "class", "path": "omniread.csv.scraper.BaseCsvClient", - "signature": "", + "signature": null, "docstring": "Abstract client responsible for retrieving csv bytes.\n\nRetrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).\n\nNotes:\n **Responsibilities:**\n\n - Implementations must accept a source identifier appropriate to\n the backing store.\n - Return the full csv binary payload.\n - Raise retrieval-specific errors on failure.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.csv.scraper.BaseCsvClient.fetch", - "signature": "", + "signature": "fetch(source: Any)", "docstring": "Fetch raw csv bytes from the given source.\n\nArgs:\n source (Any):\n Identifier of the csv location, such as a file path,\n object storage key, or remote reference.\n\nReturns:\n bytes:\n Raw csv bytes.\n\nRaises:\n Exception:\n Retrieval-specific errors defined by the implementation." } } @@ -1524,14 +1356,14 @@ "name": "CsvScraper", "kind": "class", "path": "omniread.csv.scraper.CsvScraper", - "signature": "", + "signature": "CsvScraper(*, client: BaseCsvClient)", "docstring": "Scraper for csv documents.\n\nNotes:\n **Responsibilities:**\n\n - Fetch raw csv bytes via the configured client.\n - Wrap the payload in a canonical `Content` instance with the\n CSV content type and source identifier.\n\n **Constraints:**\n\n - The scraper does not perform parsing or interpretation.\n - Does not assume a specific storage backend.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.csv.scraper.CsvScraper.fetch", - "signature": "", + "signature": "fetch(source: Any, *, metadata: Mapping[str, Any] | None = None) -> Content", "docstring": "Fetch a csv document from the given source.\n\nArgs:\n source (Any):\n Identifier of the csv source as understood by the\n configured client.\n metadata (Mapping[str, Any] | None, optional):\n Optional metadata to attach to the returned content.\n\nReturns:\n Content:\n A `Content` instance containing raw csv bytes, source\n identifier, CSV content type, and optional metadata.\n\nRaises:\n Exception:\n Retrieval-specific errors raised by the client." } } @@ -1551,28 +1383,28 @@ "name": "HTMLScraper", "kind": "class", "path": "omniread.html.HTMLScraper", - "signature": "", + "signature": "HTMLScraper(*, client: httpx.Client | None = None, timeout: float = 15.0, headers: Mapping[str, str] | None = None, follow_redirects: bool = True)", "docstring": "Base HTML scraper using `httpx`.\n\nNotes:\n **Responsibilities:**\n\n - This scraper retrieves HTML documents over HTTP(S) and returns\n them as raw content wrapped in a `Content` object.\n - Fetches raw bytes and metadata only.\n - The scraper uses `httpx.Client` for HTTP requests, enforces an\n HTML content type, and preserves HTTP response metadata.\n\n **Constraints:**\n\n - The scraper does not: Parse HTML, perform retries or backoff,\n handle non-HTML responses.", "members": { "content_type": { "name": "content_type", "kind": "attribute", "path": "omniread.html.HTMLScraper.content_type", - "signature": "", + "signature": null, "docstring": null }, "validate_content_type": { "name": "validate_content_type", "kind": "function", "path": "omniread.html.HTMLScraper.validate_content_type", - "signature": "", + "signature": "validate_content_type(response: httpx.Response)", "docstring": "Validate that the HTTP response contains HTML content.\n\nArgs:\n response (httpx.Response):\n HTTP response returned by `httpx`.\n\nRaises:\n ValueError:\n If the `Content-Type` header is missing or does not indicate HTML content." }, "fetch": { "name": "fetch", "kind": "function", "path": "omniread.html.HTMLScraper.fetch", - "signature": "", + "signature": "fetch(source: str, *, metadata: Mapping[str, Any] | None = None)", "docstring": "Fetch an HTML document from the given source.\n\nArgs:\n source (str):\n URL of the HTML document.\n metadata (Mapping[str, Any] | None, optional):\n Optional metadata to be merged into the returned content.\n\nReturns:\n Content:\n A `Content` instance containing raw HTML bytes, source URL, HTML content type, and HTTP response metadata.\n\nRaises:\n httpx.HTTPError:\n If the HTTP request fails.\n ValueError:\n If the response is not valid HTML." } } @@ -1581,49 +1413,49 @@ "name": "HTMLParser", "kind": "class", "path": "omniread.html.HTMLParser", - "signature": "", + "signature": "HTMLParser(content: Content, features: str = 'html.parser')", "docstring": "Base HTML parser.\n\nNotes:\n **Responsibilities:**\n\n - This class extends the core `BaseParser` with HTML-specific behavior,\n including DOM parsing via BeautifulSoup and reusable extraction helpers.\n - Provides reusable helpers for HTML extraction. Concrete parsers must\n explicitly define the return type.\n\n **Guarantees:**\n\n - Accepts only HTML content.\n - Owns a parsed BeautifulSoup DOM tree.\n - Provides pure helper utilities for common HTML structures.\n\n **Constraints:**\n\n - Concrete subclasses must define the output type `T` and implement\n the `parse()` method.", "members": { "supported_types": { "name": "supported_types", "kind": "attribute", "path": "omniread.html.HTMLParser.supported_types", - "signature": "", + "signature": null, "docstring": "Set of content types supported by this parser (HTML only)." }, "parse": { "name": "parse", "kind": "function", "path": "omniread.html.HTMLParser.parse", - "signature": "", + "signature": "parse()", "docstring": "Fully parse the HTML content into structured output.\n\nReturns:\n T:\n Parsed representation of type `T`.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must fully interpret the HTML DOM and return a\n deterministic, structured output." }, "parse_div": { "name": "parse_div", "kind": "function", "path": "omniread.html.HTMLParser.parse_div", - "signature": "", + "signature": "parse_div(div: Tag, *, separator: str = ' ')", "docstring": "Extract normalized text from a `
` element.\n\nArgs:\n div (Tag):\n BeautifulSoup tag representing a `
`.\n separator (str, optional):\n String used to separate text nodes.\n\nReturns:\n str:\n Flattened, whitespace-normalized text content." }, "parse_link": { "name": "parse_link", "kind": "function", "path": "omniread.html.HTMLParser.parse_link", - "signature": "", + "signature": "parse_link(a: Tag)", "docstring": "Extract the hyperlink reference from an `` element.\n\nArgs:\n a (Tag):\n BeautifulSoup tag representing an anchor.\n\nReturns:\n str | None:\n The value of the `href` attribute, or None if absent." }, "parse_table": { "name": "parse_table", "kind": "function", "path": "omniread.html.HTMLParser.parse_table", - "signature": "", + "signature": "parse_table(table: Tag)", "docstring": "Parse an HTML table into a 2D list of strings.\n\nArgs:\n table (Tag):\n BeautifulSoup tag representing a `
`.\n\nReturns:\n list[list[str]]:\n A list of rows, where each row is a list of cell text values." }, "parse_meta": { "name": "parse_meta", "kind": "function", "path": "omniread.html.HTMLParser.parse_meta", - "signature": "", + "signature": "parse_meta()", "docstring": "Extract high-level metadata from the HTML document.\n\nReturns:\n dict[str, Any]:\n Dictionary containing extracted metadata.\n\nNotes:\n **Responsibilities:**\n\n - Extract high-level metadata from the HTML document.\n - This includes: Document title, `` tag name/property to\n content mappings." } } @@ -1635,81 +1467,39 @@ "signature": null, "docstring": "# Summary\n\nHTML parser base implementations for OmniRead.\n\nThis module provides reusable HTML parsing utilities built on top of\nthe abstract parser contracts defined in `omniread.core.parser`.\n\nIt supplies:\n\n- Content-type enforcement for HTML inputs\n- BeautifulSoup initialization and lifecycle management\n- Common helper methods for extracting structured data from HTML elements\n\nConcrete parsers must subclass `HTMLParser` and implement the `parse()` method\nto return a structured representation appropriate for their use case.", "members": { - "abstractmethod": { - "name": "abstractmethod", - "kind": "alias", - "path": "omniread.html.parser.abstractmethod", - "signature": "", - "docstring": null - }, - "Any": { - "name": "Any", - "kind": "alias", - "path": "omniread.html.parser.Any", - "signature": "", - "docstring": null - }, - "Generic": { - "name": "Generic", - "kind": "alias", - "path": "omniread.html.parser.Generic", - "signature": "", - "docstring": null - }, - "TypeVar": { - "name": "TypeVar", - "kind": "alias", - "path": "omniread.html.parser.TypeVar", - "signature": "", - "docstring": null - }, - "BeautifulSoup": { - "name": "BeautifulSoup", - "kind": "alias", - "path": "omniread.html.parser.BeautifulSoup", - "signature": "", - "docstring": null - }, - "Tag": { - "name": "Tag", - "kind": "alias", - "path": "omniread.html.parser.Tag", - "signature": "", - "docstring": null - }, "Content": { "name": "Content", "kind": "class", "path": "omniread.html.parser.Content", - "signature": "", + "signature": "Content(raw: bytes, source: str, content_type: ContentType | None = ..., metadata: Mapping[str, Any] | None = ...)", "docstring": "Normalized representation of extracted content.\n\nNotes:\n **Responsibilities:**\n\n - A `Content` instance represents a raw content payload along with\n minimal contextual metadata describing its origin and type.\n - This class is the primary exchange format between scrapers,\n parsers, and downstream consumers.", "members": { "raw": { "name": "raw", "kind": "attribute", "path": "omniread.html.parser.Content.raw", - "signature": "", + "signature": null, "docstring": "Raw content bytes as retrieved from the source." }, "source": { "name": "source", "kind": "attribute", "path": "omniread.html.parser.Content.source", - "signature": "", + "signature": null, "docstring": "Identifier of the content origin (URL, file path, or logical name)." }, "content_type": { "name": "content_type", "kind": "attribute", "path": "omniread.html.parser.Content.content_type", - "signature": "", + "signature": null, "docstring": "Optional MIME type of the content, if known." }, "metadata": { "name": "metadata", "kind": "attribute", "path": "omniread.html.parser.Content.metadata", - "signature": "", + "signature": null, "docstring": "Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes)." } } @@ -1718,49 +1508,49 @@ "name": "ContentType", "kind": "class", "path": "omniread.html.parser.ContentType", - "signature": "", + "signature": null, "docstring": "Supported MIME types for extracted content.\n\nNotes:\n **Guarantees:**\n\n - This enum represents the declared or inferred media type of the\n content source.\n - It is primarily used for routing content to the appropriate\n parser or downstream consumer.", "members": { "HTML": { "name": "HTML", "kind": "attribute", "path": "omniread.html.parser.ContentType.HTML", - "signature": "", + "signature": null, "docstring": "HTML document content." }, "PDF": { "name": "PDF", "kind": "attribute", "path": "omniread.html.parser.ContentType.PDF", - "signature": "", + "signature": null, "docstring": "PDF document content." }, "XLSX": { "name": "XLSX", "kind": "attribute", "path": "omniread.html.parser.ContentType.XLSX", - "signature": "", + "signature": null, "docstring": "Office Open XML spreadsheet (xlsx/xlsm) content." }, "CSV": { "name": "CSV", "kind": "attribute", "path": "omniread.html.parser.ContentType.CSV", - "signature": "", + "signature": null, "docstring": "Comma-separated-value document content." }, "JSON": { "name": "JSON", "kind": "attribute", "path": "omniread.html.parser.ContentType.JSON", - "signature": "", + "signature": null, "docstring": "JSON document content." }, "XML": { "name": "XML", "kind": "attribute", "path": "omniread.html.parser.ContentType.XML", - "signature": "", + "signature": null, "docstring": "XML document content." } } @@ -1769,35 +1559,35 @@ "name": "BaseParser", "kind": "class", "path": "omniread.html.parser.BaseParser", - "signature": "", + "signature": "BaseParser(content: Content)", "docstring": "Base interface for all parsers.\n\nNotes:\n **Guarantees:**\n\n - A parser is a self-contained object that owns the `Content` it is\n responsible for interpreting.\n - Consumers may rely on early validation of content compatibility\n and type-stable return values from `parse()`.\n\n **Responsibilities:**\n\n - Implementations must declare supported content types via `supported_types`.\n - Implementations must raise parsing-specific exceptions from `parse()`.\n - Implementations must remain deterministic for a given input.", "members": { "supported_types": { "name": "supported_types", "kind": "attribute", "path": "omniread.html.parser.BaseParser.supported_types", - "signature": "", + "signature": null, "docstring": "Set of content types supported by this parser. An empty set indicates that the parser is content-type agnostic." }, "content": { "name": "content", "kind": "attribute", "path": "omniread.html.parser.BaseParser.content", - "signature": "", + "signature": null, "docstring": null }, "parse": { "name": "parse", "kind": "function", "path": "omniread.html.parser.BaseParser.parse", - "signature": "", + "signature": "parse()", "docstring": "Parse the owned content into structured output.\n\nReturns:\n T:\n Parsed, structured representation.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must fully consume the provided content and\n return a deterministic, structured output." }, "supports": { "name": "supports", "kind": "function", "path": "omniread.html.parser.BaseParser.supports", - "signature": "", + "signature": "supports()", "docstring": "Check whether this parser supports the content's type.\n\nReturns:\n bool:\n True if the content type is supported; False otherwise." } } @@ -1813,7 +1603,7 @@ "name": "HTMLParser", "kind": "class", "path": "omniread.html.parser.HTMLParser", - "signature": "", + "signature": "HTMLParser(content: Content, features: str = 'html.parser')", "docstring": "Base HTML parser.\n\nNotes:\n **Responsibilities:**\n\n - This class extends the core `BaseParser` with HTML-specific behavior,\n including DOM parsing via BeautifulSoup and reusable extraction helpers.\n - Provides reusable helpers for HTML extraction. Concrete parsers must\n explicitly define the return type.\n\n **Guarantees:**\n\n - Accepts only HTML content.\n - Owns a parsed BeautifulSoup DOM tree.\n - Provides pure helper utilities for common HTML structures.\n\n **Constraints:**\n\n - Concrete subclasses must define the output type `T` and implement\n the `parse()` method.", "members": { "supported_types": { @@ -1827,35 +1617,35 @@ "name": "parse", "kind": "function", "path": "omniread.html.parser.HTMLParser.parse", - "signature": "", + "signature": "parse() -> T", "docstring": "Fully parse the HTML content into structured output.\n\nReturns:\n T:\n Parsed representation of type `T`.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must fully interpret the HTML DOM and return a\n deterministic, structured output." }, "parse_div": { "name": "parse_div", "kind": "function", "path": "omniread.html.parser.HTMLParser.parse_div", - "signature": "", + "signature": "parse_div(div: Tag, *, separator: str = ' ') -> str", "docstring": "Extract normalized text from a `
` element.\n\nArgs:\n div (Tag):\n BeautifulSoup tag representing a `
`.\n separator (str, optional):\n String used to separate text nodes.\n\nReturns:\n str:\n Flattened, whitespace-normalized text content." }, "parse_link": { "name": "parse_link", "kind": "function", "path": "omniread.html.parser.HTMLParser.parse_link", - "signature": "", + "signature": "parse_link(a: Tag) -> str | None", "docstring": "Extract the hyperlink reference from an `` element.\n\nArgs:\n a (Tag):\n BeautifulSoup tag representing an anchor.\n\nReturns:\n str | None:\n The value of the `href` attribute, or None if absent." }, "parse_table": { "name": "parse_table", "kind": "function", "path": "omniread.html.parser.HTMLParser.parse_table", - "signature": "", + "signature": "parse_table(table: Tag) -> list[list[str]]", "docstring": "Parse an HTML table into a 2D list of strings.\n\nArgs:\n table (Tag):\n BeautifulSoup tag representing a `
`.\n\nReturns:\n list[list[str]]:\n A list of rows, where each row is a list of cell text values." }, "parse_meta": { "name": "parse_meta", "kind": "function", "path": "omniread.html.parser.HTMLParser.parse_meta", - "signature": "", + "signature": "parse_meta() -> dict[str, Any]", "docstring": "Extract high-level metadata from the HTML document.\n\nReturns:\n dict[str, Any]:\n Dictionary containing extracted metadata.\n\nNotes:\n **Responsibilities:**\n\n - Extract high-level metadata from the HTML document.\n - This includes: Document title, `` tag name/property to\n content mappings." } } @@ -1869,60 +1659,39 @@ "signature": null, "docstring": "# Summary\n\nHTML scraping implementation for OmniRead.\n\nThis module provides an HTTP-based scraper for retrieving HTML documents.\nIt implements the core `BaseScraper` contract using `httpx` as the transport\nlayer.\n\nThis scraper is responsible for:\n\n- Fetching raw HTML bytes over HTTP(S)\n- Validating response content type\n- Attaching HTTP metadata to the returned content\n\nThis scraper is not responsible for:\n\n- Parsing or interpreting HTML\n- Retrying failed requests\n- Managing crawl policies or rate limiting", "members": { - "Mapping": { - "name": "Mapping", - "kind": "alias", - "path": "omniread.html.scraper.Mapping", - "signature": "", - "docstring": null - }, - "Any": { - "name": "Any", - "kind": "alias", - "path": "omniread.html.scraper.Any", - "signature": "", - "docstring": null - }, - "httpx": { - "name": "httpx", - "kind": "alias", - "path": "omniread.html.scraper.httpx", - "signature": "", - "docstring": null - }, "Content": { "name": "Content", "kind": "class", "path": "omniread.html.scraper.Content", - "signature": "", + "signature": "Content(raw: bytes, source: str, content_type: ContentType | None = ..., metadata: Mapping[str, Any] | None = ...)", "docstring": "Normalized representation of extracted content.\n\nNotes:\n **Responsibilities:**\n\n - A `Content` instance represents a raw content payload along with\n minimal contextual metadata describing its origin and type.\n - This class is the primary exchange format between scrapers,\n parsers, and downstream consumers.", "members": { "raw": { "name": "raw", "kind": "attribute", "path": "omniread.html.scraper.Content.raw", - "signature": "", + "signature": null, "docstring": "Raw content bytes as retrieved from the source." }, "source": { "name": "source", "kind": "attribute", "path": "omniread.html.scraper.Content.source", - "signature": "", + "signature": null, "docstring": "Identifier of the content origin (URL, file path, or logical name)." }, "content_type": { "name": "content_type", "kind": "attribute", "path": "omniread.html.scraper.Content.content_type", - "signature": "", + "signature": null, "docstring": "Optional MIME type of the content, if known." }, "metadata": { "name": "metadata", "kind": "attribute", "path": "omniread.html.scraper.Content.metadata", - "signature": "", + "signature": null, "docstring": "Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes)." } } @@ -1931,49 +1700,49 @@ "name": "ContentType", "kind": "class", "path": "omniread.html.scraper.ContentType", - "signature": "", + "signature": null, "docstring": "Supported MIME types for extracted content.\n\nNotes:\n **Guarantees:**\n\n - This enum represents the declared or inferred media type of the\n content source.\n - It is primarily used for routing content to the appropriate\n parser or downstream consumer.", "members": { "HTML": { "name": "HTML", "kind": "attribute", "path": "omniread.html.scraper.ContentType.HTML", - "signature": "", + "signature": null, "docstring": "HTML document content." }, "PDF": { "name": "PDF", "kind": "attribute", "path": "omniread.html.scraper.ContentType.PDF", - "signature": "", + "signature": null, "docstring": "PDF document content." }, "XLSX": { "name": "XLSX", "kind": "attribute", "path": "omniread.html.scraper.ContentType.XLSX", - "signature": "", + "signature": null, "docstring": "Office Open XML spreadsheet (xlsx/xlsm) content." }, "CSV": { "name": "CSV", "kind": "attribute", "path": "omniread.html.scraper.ContentType.CSV", - "signature": "", + "signature": null, "docstring": "Comma-separated-value document content." }, "JSON": { "name": "JSON", "kind": "attribute", "path": "omniread.html.scraper.ContentType.JSON", - "signature": "", + "signature": null, "docstring": "JSON document content." }, "XML": { "name": "XML", "kind": "attribute", "path": "omniread.html.scraper.ContentType.XML", - "signature": "", + "signature": null, "docstring": "XML document content." } } @@ -1982,14 +1751,14 @@ "name": "BaseScraper", "kind": "class", "path": "omniread.html.scraper.BaseScraper", - "signature": "", + "signature": null, "docstring": "Base interface for all scrapers.\n\nNotes:\n **Responsibilities:**\n\n - A scraper is responsible ONLY for fetching raw content (bytes)\n from a source. It must not interpret or parse it.\n - A scraper is a stateless acquisition component that retrieves raw\n content from a source and returns it as a `Content` object.\n - Scrapers define how content is obtained, not what the content means.\n - Implementations may vary in transport mechanism, authentication\n strategy, retry and backoff behavior.\n\n **Constraints:**\n\n - Implementations must not parse content, modify content semantics,\n or couple scraping logic to a specific parser.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.html.scraper.BaseScraper.fetch", - "signature": "", + "signature": "fetch(source: str, *, metadata: Mapping[str, Any] | None = None)", "docstring": "Fetch raw content from the given source.\n\nArgs:\n source (str):\n Location identifier (URL, file path, S3 URI, etc.).\n\n metadata (Mapping[str, Any] | None, optional):\n Optional hints for the scraper (headers, auth, etc.).\n\nReturns:\n Content:\n Content object containing raw bytes and metadata.\n\nRaises:\n Exception:\n Retrieval-specific errors as defined by the implementation.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must retrieve the content referenced by `source`\n and return it as raw bytes wrapped in a `Content` object." } } @@ -1998,7 +1767,7 @@ "name": "HTMLScraper", "kind": "class", "path": "omniread.html.scraper.HTMLScraper", - "signature": "", + "signature": "HTMLScraper(*, client: httpx.Client | None = None, timeout: float = 15.0, headers: Mapping[str, str] | None = None, follow_redirects: bool = True)", "docstring": "Base HTML scraper using `httpx`.\n\nNotes:\n **Responsibilities:**\n\n - This scraper retrieves HTML documents over HTTP(S) and returns\n them as raw content wrapped in a `Content` object.\n - Fetches raw bytes and metadata only.\n - The scraper uses `httpx.Client` for HTTP requests, enforces an\n HTML content type, and preserves HTTP response metadata.\n\n **Constraints:**\n\n - The scraper does not: Parse HTML, perform retries or backoff,\n handle non-HTML responses.", "members": { "content_type": { @@ -2012,14 +1781,14 @@ "name": "validate_content_type", "kind": "function", "path": "omniread.html.scraper.HTMLScraper.validate_content_type", - "signature": "", + "signature": "validate_content_type(response: httpx.Response) -> None", "docstring": "Validate that the HTTP response contains HTML content.\n\nArgs:\n response (httpx.Response):\n HTTP response returned by `httpx`.\n\nRaises:\n ValueError:\n If the `Content-Type` header is missing or does not indicate HTML content." }, "fetch": { "name": "fetch", "kind": "function", "path": "omniread.html.scraper.HTMLScraper.fetch", - "signature": "", + "signature": "fetch(source: str, *, metadata: Mapping[str, Any] | None = None) -> Content", "docstring": "Fetch an HTML document from the given source.\n\nArgs:\n source (str):\n URL of the HTML document.\n metadata (Mapping[str, Any] | None, optional):\n Optional metadata to be merged into the returned content.\n\nReturns:\n Content:\n A `Content` instance containing raw HTML bytes, source URL, HTML content type, and HTTP response metadata.\n\nRaises:\n httpx.HTTPError:\n If the HTTP request fails.\n ValueError:\n If the response is not valid HTML." } } @@ -2039,14 +1808,14 @@ "name": "FileSystemPDFClient", "kind": "class", "path": "omniread.pdf.FileSystemPDFClient", - "signature": "", + "signature": null, "docstring": "PDF client that reads from the local filesystem.\n\nNotes:\n **Guarantees:**\n\n - This client reads PDF files directly from the disk and returns\n their raw binary contents.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.pdf.FileSystemPDFClient.fetch", - "signature": "", + "signature": "fetch(path: Path)", "docstring": "Read a PDF file from the local filesystem.\n\nArgs:\n path (Path):\n Filesystem path to the PDF file.\n\nReturns:\n bytes:\n Raw PDF bytes.\n\nRaises:\n FileNotFoundError:\n If the path does not exist.\n ValueError:\n If the path exists but is not a file." } } @@ -2055,14 +1824,14 @@ "name": "PDFScraper", "kind": "class", "path": "omniread.pdf.PDFScraper", - "signature": "", + "signature": "PDFScraper(*, client: BasePDFClient)", "docstring": "Scraper for PDF sources.\n\nNotes:\n **Responsibilities:**\n\n - Delegates byte retrieval to a PDF client and normalizes output\n into `Content`.\n - Preserves caller-provided metadata.\n\n **Constraints:**\n\n - The scraper does not perform parsing or interpretation.\n - Does not assume a specific storage backend.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.pdf.PDFScraper.fetch", - "signature": "", + "signature": "fetch(source: Any, *, metadata: Mapping[str, Any] | None = None)", "docstring": "Fetch a PDF document from the given source.\n\nArgs:\n source (Any):\n Identifier of the PDF source as understood by the configured PDF client.\n metadata (Mapping[str, Any] | None, optional):\n Optional metadata to attach to the returned content.\n\nReturns:\n Content:\n A `Content` instance containing raw PDF bytes, source identifier, PDF content type, and optional metadata.\n\nRaises:\n Exception:\n Retrieval-specific errors raised by the PDF client." } } @@ -2071,21 +1840,21 @@ "name": "PDFParser", "kind": "class", "path": "omniread.pdf.PDFParser", - "signature": "", + "signature": null, "docstring": "Base PDF parser.\n\nNotes:\n **Responsibilities:**\n\n - This class enforces PDF content-type compatibility and provides\n the extension point for implementing concrete PDF parsing strategies.\n\n **Constraints:**\n\n - Concrete implementations must define the output type `T` and\n implement the `parse()` method.", "members": { "supported_types": { "name": "supported_types", "kind": "attribute", "path": "omniread.pdf.PDFParser.supported_types", - "signature": "", + "signature": null, "docstring": "Set of content types supported by this parser (PDF only)." }, "parse": { "name": "parse", "kind": "function", "path": "omniread.pdf.PDFParser.parse", - "signature": "", + "signature": "parse()", "docstring": "Parse PDF content into a structured output.\n\nReturns:\n T:\n Parsed representation of type `T`.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must fully interpret the PDF binary payload and\n return a deterministic, structured output." } } @@ -2097,46 +1866,18 @@ "signature": null, "docstring": "# Summary\n\nPDF client abstractions for OmniRead.\n\nThis module defines the **client layer** responsible for retrieving raw PDF\nbytes from a concrete backing store.\n\nClients provide low-level access to PDF binaries and are intentionally\ndecoupled from scraping and parsing logic. They do not perform validation,\ninterpretation, or content extraction.\n\nTypical backing stores include:\n\n- Local filesystems\n- Object storage (S3, GCS, etc.)\n- Network file systems", "members": { - "ABC": { - "name": "ABC", - "kind": "alias", - "path": "omniread.pdf.client.ABC", - "signature": "", - "docstring": null - }, - "abstractmethod": { - "name": "abstractmethod", - "kind": "alias", - "path": "omniread.pdf.client.abstractmethod", - "signature": "", - "docstring": null - }, - "Path": { - "name": "Path", - "kind": "alias", - "path": "omniread.pdf.client.Path", - "signature": "", - "docstring": null - }, - "Any": { - "name": "Any", - "kind": "alias", - "path": "omniread.pdf.client.Any", - "signature": "", - "docstring": null - }, "BasePDFClient": { "name": "BasePDFClient", "kind": "class", "path": "omniread.pdf.client.BasePDFClient", - "signature": "", + "signature": null, "docstring": "Abstract client responsible for retrieving PDF bytes.\n\nRetrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).\n\nNotes:\n **Responsibilities:**\n\n - Implementations must accept a source identifier appropriate to\n the backing store.\n - Return the full PDF binary payload.\n - Raise retrieval-specific errors on failure.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.pdf.client.BasePDFClient.fetch", - "signature": "", + "signature": "fetch(source: Any) -> bytes", "docstring": "Fetch raw PDF bytes from the given source.\n\nArgs:\n source (Any):\n Identifier of the PDF location, such as a file path, object storage key, or remote reference.\n\nReturns:\n bytes:\n Raw PDF bytes.\n\nRaises:\n Exception:\n Retrieval-specific errors defined by the implementation." } } @@ -2145,14 +1886,14 @@ "name": "FileSystemPDFClient", "kind": "class", "path": "omniread.pdf.client.FileSystemPDFClient", - "signature": "", + "signature": null, "docstring": "PDF client that reads from the local filesystem.\n\nNotes:\n **Guarantees:**\n\n - This client reads PDF files directly from the disk and returns\n their raw binary contents.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.pdf.client.FileSystemPDFClient.fetch", - "signature": "", + "signature": "fetch(path: Path) -> bytes", "docstring": "Read a PDF file from the local filesystem.\n\nArgs:\n path (Path):\n Filesystem path to the PDF file.\n\nReturns:\n bytes:\n Raw PDF bytes.\n\nRaises:\n FileNotFoundError:\n If the path does not exist.\n ValueError:\n If the path exists but is not a file." } } @@ -2166,74 +1907,53 @@ "signature": null, "docstring": "# Summary\n\nPDF parser base implementations for OmniRead.\n\nThis module defines the **PDF-specific parser contract**, extending the\nformat-agnostic `BaseParser` with constraints appropriate for PDF content.\n\nPDF parsers are responsible for interpreting binary PDF data and producing\nstructured representations suitable for downstream consumption.", "members": { - "abstractmethod": { - "name": "abstractmethod", - "kind": "alias", - "path": "omniread.pdf.parser.abstractmethod", - "signature": "", - "docstring": null - }, - "Generic": { - "name": "Generic", - "kind": "alias", - "path": "omniread.pdf.parser.Generic", - "signature": "", - "docstring": null - }, - "TypeVar": { - "name": "TypeVar", - "kind": "alias", - "path": "omniread.pdf.parser.TypeVar", - "signature": "", - "docstring": null - }, "ContentType": { "name": "ContentType", "kind": "class", "path": "omniread.pdf.parser.ContentType", - "signature": "", + "signature": null, "docstring": "Supported MIME types for extracted content.\n\nNotes:\n **Guarantees:**\n\n - This enum represents the declared or inferred media type of the\n content source.\n - It is primarily used for routing content to the appropriate\n parser or downstream consumer.", "members": { "HTML": { "name": "HTML", "kind": "attribute", "path": "omniread.pdf.parser.ContentType.HTML", - "signature": "", + "signature": null, "docstring": "HTML document content." }, "PDF": { "name": "PDF", "kind": "attribute", "path": "omniread.pdf.parser.ContentType.PDF", - "signature": "", + "signature": null, "docstring": "PDF document content." }, "XLSX": { "name": "XLSX", "kind": "attribute", "path": "omniread.pdf.parser.ContentType.XLSX", - "signature": "", + "signature": null, "docstring": "Office Open XML spreadsheet (xlsx/xlsm) content." }, "CSV": { "name": "CSV", "kind": "attribute", "path": "omniread.pdf.parser.ContentType.CSV", - "signature": "", + "signature": null, "docstring": "Comma-separated-value document content." }, "JSON": { "name": "JSON", "kind": "attribute", "path": "omniread.pdf.parser.ContentType.JSON", - "signature": "", + "signature": null, "docstring": "JSON document content." }, "XML": { "name": "XML", "kind": "attribute", "path": "omniread.pdf.parser.ContentType.XML", - "signature": "", + "signature": null, "docstring": "XML document content." } } @@ -2242,35 +1962,35 @@ "name": "BaseParser", "kind": "class", "path": "omniread.pdf.parser.BaseParser", - "signature": "", + "signature": "BaseParser(content: Content)", "docstring": "Base interface for all parsers.\n\nNotes:\n **Guarantees:**\n\n - A parser is a self-contained object that owns the `Content` it is\n responsible for interpreting.\n - Consumers may rely on early validation of content compatibility\n and type-stable return values from `parse()`.\n\n **Responsibilities:**\n\n - Implementations must declare supported content types via `supported_types`.\n - Implementations must raise parsing-specific exceptions from `parse()`.\n - Implementations must remain deterministic for a given input.", "members": { "supported_types": { "name": "supported_types", "kind": "attribute", "path": "omniread.pdf.parser.BaseParser.supported_types", - "signature": "", + "signature": null, "docstring": "Set of content types supported by this parser. An empty set indicates that the parser is content-type agnostic." }, "content": { "name": "content", "kind": "attribute", "path": "omniread.pdf.parser.BaseParser.content", - "signature": "", + "signature": null, "docstring": null }, "parse": { "name": "parse", "kind": "function", "path": "omniread.pdf.parser.BaseParser.parse", - "signature": "", + "signature": "parse()", "docstring": "Parse the owned content into structured output.\n\nReturns:\n T:\n Parsed, structured representation.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must fully consume the provided content and\n return a deterministic, structured output." }, "supports": { "name": "supports", "kind": "function", "path": "omniread.pdf.parser.BaseParser.supports", - "signature": "", + "signature": "supports()", "docstring": "Check whether this parser supports the content's type.\n\nReturns:\n bool:\n True if the content type is supported; False otherwise." } } @@ -2286,7 +2006,7 @@ "name": "PDFParser", "kind": "class", "path": "omniread.pdf.parser.PDFParser", - "signature": "", + "signature": null, "docstring": "Base PDF parser.\n\nNotes:\n **Responsibilities:**\n\n - This class enforces PDF content-type compatibility and provides\n the extension point for implementing concrete PDF parsing strategies.\n\n **Constraints:**\n\n - Concrete implementations must define the output type `T` and\n implement the `parse()` method.", "members": { "supported_types": { @@ -2300,7 +2020,7 @@ "name": "parse", "kind": "function", "path": "omniread.pdf.parser.PDFParser.parse", - "signature": "", + "signature": "parse() -> T", "docstring": "Parse PDF content into a structured output.\n\nReturns:\n T:\n Parsed representation of type `T`.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must fully interpret the PDF binary payload and\n return a deterministic, structured output." } } @@ -2314,53 +2034,39 @@ "signature": null, "docstring": "# Summary\n\nPDF scraping implementation for OmniRead.\n\nThis module provides a PDF-specific scraper that coordinates PDF byte\nretrieval via a client and normalizes the result into a `Content` object.\n\nThe scraper implements the core `BaseScraper` contract while delegating\nall storage and access concerns to a `BasePDFClient` implementation.", "members": { - "Mapping": { - "name": "Mapping", - "kind": "alias", - "path": "omniread.pdf.scraper.Mapping", - "signature": "", - "docstring": null - }, - "Any": { - "name": "Any", - "kind": "alias", - "path": "omniread.pdf.scraper.Any", - "signature": "", - "docstring": null - }, "Content": { "name": "Content", "kind": "class", "path": "omniread.pdf.scraper.Content", - "signature": "", + "signature": "Content(raw: bytes, source: str, content_type: ContentType | None = ..., metadata: Mapping[str, Any] | None = ...)", "docstring": "Normalized representation of extracted content.\n\nNotes:\n **Responsibilities:**\n\n - A `Content` instance represents a raw content payload along with\n minimal contextual metadata describing its origin and type.\n - This class is the primary exchange format between scrapers,\n parsers, and downstream consumers.", "members": { "raw": { "name": "raw", "kind": "attribute", "path": "omniread.pdf.scraper.Content.raw", - "signature": "", + "signature": null, "docstring": "Raw content bytes as retrieved from the source." }, "source": { "name": "source", "kind": "attribute", "path": "omniread.pdf.scraper.Content.source", - "signature": "", + "signature": null, "docstring": "Identifier of the content origin (URL, file path, or logical name)." }, "content_type": { "name": "content_type", "kind": "attribute", "path": "omniread.pdf.scraper.Content.content_type", - "signature": "", + "signature": null, "docstring": "Optional MIME type of the content, if known." }, "metadata": { "name": "metadata", "kind": "attribute", "path": "omniread.pdf.scraper.Content.metadata", - "signature": "", + "signature": null, "docstring": "Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes)." } } @@ -2369,49 +2075,49 @@ "name": "ContentType", "kind": "class", "path": "omniread.pdf.scraper.ContentType", - "signature": "", + "signature": null, "docstring": "Supported MIME types for extracted content.\n\nNotes:\n **Guarantees:**\n\n - This enum represents the declared or inferred media type of the\n content source.\n - It is primarily used for routing content to the appropriate\n parser or downstream consumer.", "members": { "HTML": { "name": "HTML", "kind": "attribute", "path": "omniread.pdf.scraper.ContentType.HTML", - "signature": "", + "signature": null, "docstring": "HTML document content." }, "PDF": { "name": "PDF", "kind": "attribute", "path": "omniread.pdf.scraper.ContentType.PDF", - "signature": "", + "signature": null, "docstring": "PDF document content." }, "XLSX": { "name": "XLSX", "kind": "attribute", "path": "omniread.pdf.scraper.ContentType.XLSX", - "signature": "", + "signature": null, "docstring": "Office Open XML spreadsheet (xlsx/xlsm) content." }, "CSV": { "name": "CSV", "kind": "attribute", "path": "omniread.pdf.scraper.ContentType.CSV", - "signature": "", + "signature": null, "docstring": "Comma-separated-value document content." }, "JSON": { "name": "JSON", "kind": "attribute", "path": "omniread.pdf.scraper.ContentType.JSON", - "signature": "", + "signature": null, "docstring": "JSON document content." }, "XML": { "name": "XML", "kind": "attribute", "path": "omniread.pdf.scraper.ContentType.XML", - "signature": "", + "signature": null, "docstring": "XML document content." } } @@ -2420,14 +2126,14 @@ "name": "BaseScraper", "kind": "class", "path": "omniread.pdf.scraper.BaseScraper", - "signature": "", + "signature": null, "docstring": "Base interface for all scrapers.\n\nNotes:\n **Responsibilities:**\n\n - A scraper is responsible ONLY for fetching raw content (bytes)\n from a source. It must not interpret or parse it.\n - A scraper is a stateless acquisition component that retrieves raw\n content from a source and returns it as a `Content` object.\n - Scrapers define how content is obtained, not what the content means.\n - Implementations may vary in transport mechanism, authentication\n strategy, retry and backoff behavior.\n\n **Constraints:**\n\n - Implementations must not parse content, modify content semantics,\n or couple scraping logic to a specific parser.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.pdf.scraper.BaseScraper.fetch", - "signature": "", + "signature": "fetch(source: str, *, metadata: Mapping[str, Any] | None = None)", "docstring": "Fetch raw content from the given source.\n\nArgs:\n source (str):\n Location identifier (URL, file path, S3 URI, etc.).\n\n metadata (Mapping[str, Any] | None, optional):\n Optional hints for the scraper (headers, auth, etc.).\n\nReturns:\n Content:\n Content object containing raw bytes and metadata.\n\nRaises:\n Exception:\n Retrieval-specific errors as defined by the implementation.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must retrieve the content referenced by `source`\n and return it as raw bytes wrapped in a `Content` object." } } @@ -2436,14 +2142,14 @@ "name": "BasePDFClient", "kind": "class", "path": "omniread.pdf.scraper.BasePDFClient", - "signature": "", + "signature": null, "docstring": "Abstract client responsible for retrieving PDF bytes.\n\nRetrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).\n\nNotes:\n **Responsibilities:**\n\n - Implementations must accept a source identifier appropriate to\n the backing store.\n - Return the full PDF binary payload.\n - Raise retrieval-specific errors on failure.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.pdf.scraper.BasePDFClient.fetch", - "signature": "", + "signature": "fetch(source: Any)", "docstring": "Fetch raw PDF bytes from the given source.\n\nArgs:\n source (Any):\n Identifier of the PDF location, such as a file path, object storage key, or remote reference.\n\nReturns:\n bytes:\n Raw PDF bytes.\n\nRaises:\n Exception:\n Retrieval-specific errors defined by the implementation." } } @@ -2452,14 +2158,14 @@ "name": "PDFScraper", "kind": "class", "path": "omniread.pdf.scraper.PDFScraper", - "signature": "", + "signature": "PDFScraper(*, client: BasePDFClient)", "docstring": "Scraper for PDF sources.\n\nNotes:\n **Responsibilities:**\n\n - Delegates byte retrieval to a PDF client and normalizes output\n into `Content`.\n - Preserves caller-provided metadata.\n\n **Constraints:**\n\n - The scraper does not perform parsing or interpretation.\n - Does not assume a specific storage backend.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.pdf.scraper.PDFScraper.fetch", - "signature": "", + "signature": "fetch(source: Any, *, metadata: Mapping[str, Any] | None = None) -> Content", "docstring": "Fetch a PDF document from the given source.\n\nArgs:\n source (Any):\n Identifier of the PDF source as understood by the configured PDF client.\n metadata (Mapping[str, Any] | None, optional):\n Optional metadata to attach to the returned content.\n\nReturns:\n Content:\n A `Content` instance containing raw PDF bytes, source identifier, PDF content type, and optional metadata.\n\nRaises:\n Exception:\n Retrieval-specific errors raised by the PDF client." } } @@ -2479,14 +2185,14 @@ "name": "BaseXlsxClient", "kind": "class", "path": "omniread.xlsx.BaseXlsxClient", - "signature": "", + "signature": null, "docstring": "Abstract client responsible for retrieving spreadsheet bytes.\n\nRetrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).\n\nNotes:\n **Responsibilities:**\n\n - Implementations must accept a source identifier appropriate to\n the backing store.\n - Return the full xlsx binary payload.\n - Raise retrieval-specific errors on failure.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.xlsx.BaseXlsxClient.fetch", - "signature": "", + "signature": "fetch(source: Any)", "docstring": "Fetch raw xlsx bytes from the given source.\n\nArgs:\n source (Any):\n Identifier of the spreadsheet location, such as a file path,\n object storage key, or remote reference.\n\nReturns:\n bytes:\n Raw xlsx bytes.\n\nRaises:\n Exception:\n Retrieval-specific errors defined by the implementation." } } @@ -2495,14 +2201,14 @@ "name": "FileSystemXlsxClient", "kind": "class", "path": "omniread.xlsx.FileSystemXlsxClient", - "signature": "", + "signature": null, "docstring": "XLSX client that reads from the local filesystem.\n\nNotes:\n **Guarantees:**\n\n - This client reads spreadsheet files directly from the disk and\n returns their raw binary contents.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.xlsx.FileSystemXlsxClient.fetch", - "signature": "", + "signature": "fetch(path: Path)", "docstring": "Read an xlsx file from the local filesystem.\n\nArgs:\n path (Path):\n Filesystem path to the spreadsheet file.\n\nReturns:\n bytes:\n Raw xlsx bytes.\n\nRaises:\n FileNotFoundError:\n If the path does not exist.\n ValueError:\n If the path exists but is not a file." } } @@ -2511,35 +2217,35 @@ "name": "XlsxParser", "kind": "class", "path": "omniread.xlsx.XlsxParser", - "signature": "", + "signature": "XlsxParser(content: Content, *, data_only: bool = True, read_only: bool = True)", "docstring": "Generic xlsx parser producing string rows from a worksheet.\n\nNotes:\n **Responsibilities:**\n\n - Lazily load the workbook owned by the parser's content.\n - Normalize cells (including dates and numeric values) into\n deterministic string representations.\n - Expose sheet discovery and row extraction helpers.\n\n **Constraints:**\n\n - Cells are rendered with ``str(value)`` after trimming; date and\n datetime values are rendered in ISO format. Consumers requiring\n locale-specific formatting must convert on their side.", "members": { "workbook": { "name": "workbook", "kind": "attribute", "path": "omniread.xlsx.XlsxParser.workbook", - "signature": "", + "signature": null, "docstring": "The lazily loaded workbook backing this parser's content." }, "sheet_names": { "name": "sheet_names", "kind": "attribute", "path": "omniread.xlsx.XlsxParser.sheet_names", - "signature": "", + "signature": null, "docstring": "Names of all worksheets contained in the workbook." }, "parse": { "name": "parse", "kind": "function", "path": "omniread.xlsx.XlsxParser.parse", - "signature": "", + "signature": "parse()", "docstring": "Parse the first worksheet into normalized string rows.\n\nReturns:\n list[list[str]]:\n Rows of the default (first) worksheet." }, "rows": { "name": "rows", "kind": "function", "path": "omniread.xlsx.XlsxParser.rows", - "signature": "", + "signature": "rows(sheet: int | str | None = None, *, skip_empty: bool = True)", "docstring": "Extract normalized string rows from a worksheet.\n\nArgs:\n sheet (int | str | None):\n Worksheet index or title; defaults to the first worksheet.\n skip_empty (bool):\n When True (default), rows whose cells are all blank are omitted.\n\nReturns:\n list[list[str]]:\n Normalized rows; trailing blank cells are trimmed per row.\n\nRaises:\n ValueError:\n If the requested sheet does not exist." } } @@ -2548,21 +2254,21 @@ "name": "XlsxParserBase", "kind": "class", "path": "omniread.xlsx.XlsxParserBase", - "signature": "", + "signature": null, "docstring": "Base xlsx parser.\n\nNotes:\n **Responsibilities:**\n\n - This class enforces xlsx content-type compatibility and provides\n the extension point for implementing concrete xlsx parsing\n strategies.\n\n **Constraints:**\n\n - Concrete implementations must define the output type `T` and\n implement the `parse()` method.", "members": { "supported_types": { "name": "supported_types", "kind": "attribute", "path": "omniread.xlsx.XlsxParserBase.supported_types", - "signature": "", + "signature": null, "docstring": "Set of content types supported by this parser (XLSX only)." }, "parse": { "name": "parse", "kind": "function", "path": "omniread.xlsx.XlsxParserBase.parse", - "signature": "", + "signature": "parse()", "docstring": "Parse xlsx content into a structured output.\n\nReturns:\n T:\n Parsed representation of type `T`.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation." } } @@ -2571,14 +2277,14 @@ "name": "XlsxScraper", "kind": "class", "path": "omniread.xlsx.XlsxScraper", - "signature": "", + "signature": "XlsxScraper(*, client: BaseXlsxClient)", "docstring": "Scraper for xlsx spreadsheet documents.\n\nNotes:\n **Responsibilities:**\n\n - Fetch raw xlsx bytes via the configured client.\n - Wrap the payload in a canonical `Content` instance with the\n XLSX content type and source identifier.\n\n **Constraints:**\n\n - The scraper does not perform parsing or interpretation.\n - Does not assume a specific storage backend.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.xlsx.XlsxScraper.fetch", - "signature": "", + "signature": "fetch(source: Any, *, metadata: Mapping[str, Any] | None = None)", "docstring": "Fetch an xlsx document from the given source.\n\nArgs:\n source (Any):\n Identifier of the spreadsheet source as understood by the\n configured client.\n metadata (Mapping[str, Any] | None, optional):\n Optional metadata to attach to the returned content.\n\nReturns:\n Content:\n A `Content` instance containing raw xlsx bytes, source\n identifier, XLSX content type, and optional metadata.\n\nRaises:\n Exception:\n Retrieval-specific errors raised by the client." } } @@ -2590,46 +2296,18 @@ "signature": null, "docstring": "# Summary\n\nXLSX client abstractions for OmniRead.\n\nThis module defines the **client layer** responsible for retrieving raw\nOffice Open XML spreadsheet bytes from a concrete backing store.\n\nClients provide low-level access to xlsx binaries and are intentionally\ndecoupled from scraping and parsing logic. They do not perform validation,\ninterpretation, or content extraction.\n\nTypical backing stores include:\n\n- Local filesystems\n- Object storage (S3, GCS, etc.)\n- Network file systems", "members": { - "ABC": { - "name": "ABC", - "kind": "alias", - "path": "omniread.xlsx.client.ABC", - "signature": "", - "docstring": null - }, - "abstractmethod": { - "name": "abstractmethod", - "kind": "alias", - "path": "omniread.xlsx.client.abstractmethod", - "signature": "", - "docstring": null - }, - "Path": { - "name": "Path", - "kind": "alias", - "path": "omniread.xlsx.client.Path", - "signature": "", - "docstring": null - }, - "Any": { - "name": "Any", - "kind": "alias", - "path": "omniread.xlsx.client.Any", - "signature": "", - "docstring": null - }, "BaseXlsxClient": { "name": "BaseXlsxClient", "kind": "class", "path": "omniread.xlsx.client.BaseXlsxClient", - "signature": "", + "signature": null, "docstring": "Abstract client responsible for retrieving spreadsheet bytes.\n\nRetrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).\n\nNotes:\n **Responsibilities:**\n\n - Implementations must accept a source identifier appropriate to\n the backing store.\n - Return the full xlsx binary payload.\n - Raise retrieval-specific errors on failure.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.xlsx.client.BaseXlsxClient.fetch", - "signature": "", + "signature": "fetch(source: Any) -> bytes", "docstring": "Fetch raw xlsx bytes from the given source.\n\nArgs:\n source (Any):\n Identifier of the spreadsheet location, such as a file path,\n object storage key, or remote reference.\n\nReturns:\n bytes:\n Raw xlsx bytes.\n\nRaises:\n Exception:\n Retrieval-specific errors defined by the implementation." } } @@ -2638,14 +2316,14 @@ "name": "FileSystemXlsxClient", "kind": "class", "path": "omniread.xlsx.client.FileSystemXlsxClient", - "signature": "", + "signature": null, "docstring": "XLSX client that reads from the local filesystem.\n\nNotes:\n **Guarantees:**\n\n - This client reads spreadsheet files directly from the disk and\n returns their raw binary contents.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.xlsx.client.FileSystemXlsxClient.fetch", - "signature": "", + "signature": "fetch(path: Path) -> bytes", "docstring": "Read an xlsx file from the local filesystem.\n\nArgs:\n path (Path):\n Filesystem path to the spreadsheet file.\n\nReturns:\n bytes:\n Raw xlsx bytes.\n\nRaises:\n FileNotFoundError:\n If the path does not exist.\n ValueError:\n If the path exists but is not a file." } } @@ -2659,67 +2337,39 @@ "signature": null, "docstring": "# Summary\n\nXLSX parser implementations for OmniRead.\n\nThis module provides a concrete, generic parser for Office Open XML\nspreadsheets. It exposes workbook sheets as lists of string rows so\ndownstream consumers can interpret tabular content without depending on\nopenpyxl directly.\n\nThe parser is intentionally statement-agnostic: it performs no header\ndetection or column interpretation beyond basic cell normalization.", "members": { - "datetime": { - "name": "datetime", - "kind": "alias", - "path": "omniread.xlsx.parser.datetime", - "signature": "", - "docstring": null - }, - "BytesIO": { - "name": "BytesIO", - "kind": "alias", - "path": "omniread.xlsx.parser.BytesIO", - "signature": "", - "docstring": null - }, - "Any": { - "name": "Any", - "kind": "alias", - "path": "omniread.xlsx.parser.Any", - "signature": "", - "docstring": null - }, - "openpyxl": { - "name": "openpyxl", - "kind": "alias", - "path": "omniread.xlsx.parser.openpyxl", - "signature": "", - "docstring": null - }, "Content": { "name": "Content", "kind": "class", "path": "omniread.xlsx.parser.Content", - "signature": "", + "signature": "Content(raw: bytes, source: str, content_type: ContentType | None = ..., metadata: Mapping[str, Any] | None = ...)", "docstring": "Normalized representation of extracted content.\n\nNotes:\n **Responsibilities:**\n\n - A `Content` instance represents a raw content payload along with\n minimal contextual metadata describing its origin and type.\n - This class is the primary exchange format between scrapers,\n parsers, and downstream consumers.", "members": { "raw": { "name": "raw", "kind": "attribute", "path": "omniread.xlsx.parser.Content.raw", - "signature": "", + "signature": null, "docstring": "Raw content bytes as retrieved from the source." }, "source": { "name": "source", "kind": "attribute", "path": "omniread.xlsx.parser.Content.source", - "signature": "", + "signature": null, "docstring": "Identifier of the content origin (URL, file path, or logical name)." }, "content_type": { "name": "content_type", "kind": "attribute", "path": "omniread.xlsx.parser.Content.content_type", - "signature": "", + "signature": null, "docstring": "Optional MIME type of the content, if known." }, "metadata": { "name": "metadata", "kind": "attribute", "path": "omniread.xlsx.parser.Content.metadata", - "signature": "", + "signature": null, "docstring": "Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes)." } } @@ -2728,21 +2378,21 @@ "name": "XlsxParserBase", "kind": "class", "path": "omniread.xlsx.parser.XlsxParserBase", - "signature": "", + "signature": null, "docstring": "Base xlsx parser.\n\nNotes:\n **Responsibilities:**\n\n - This class enforces xlsx content-type compatibility and provides\n the extension point for implementing concrete xlsx parsing\n strategies.\n\n **Constraints:**\n\n - Concrete implementations must define the output type `T` and\n implement the `parse()` method.", "members": { "supported_types": { "name": "supported_types", "kind": "attribute", "path": "omniread.xlsx.parser.XlsxParserBase.supported_types", - "signature": "", + "signature": null, "docstring": "Set of content types supported by this parser (XLSX only)." }, "parse": { "name": "parse", "kind": "function", "path": "omniread.xlsx.parser.XlsxParserBase.parse", - "signature": "", + "signature": "parse()", "docstring": "Parse xlsx content into a structured output.\n\nReturns:\n T:\n Parsed representation of type `T`.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation." } } @@ -2751,7 +2401,7 @@ "name": "XlsxParser", "kind": "class", "path": "omniread.xlsx.parser.XlsxParser", - "signature": "", + "signature": "XlsxParser(content: Content, *, data_only: bool = True, read_only: bool = True)", "docstring": "Generic xlsx parser producing string rows from a worksheet.\n\nNotes:\n **Responsibilities:**\n\n - Lazily load the workbook owned by the parser's content.\n - Normalize cells (including dates and numeric values) into\n deterministic string representations.\n - Expose sheet discovery and row extraction helpers.\n\n **Constraints:**\n\n - Cells are rendered with ``str(value)`` after trimming; date and\n datetime values are rendered in ISO format. Consumers requiring\n locale-specific formatting must convert on their side.", "members": { "workbook": { @@ -2772,14 +2422,14 @@ "name": "parse", "kind": "function", "path": "omniread.xlsx.parser.XlsxParser.parse", - "signature": "", + "signature": "parse() -> list[list[str]]", "docstring": "Parse the first worksheet into normalized string rows.\n\nReturns:\n list[list[str]]:\n Rows of the default (first) worksheet." }, "rows": { "name": "rows", "kind": "function", "path": "omniread.xlsx.parser.XlsxParser.rows", - "signature": "", + "signature": "rows(sheet: int | str | None = None, *, skip_empty: bool = True) -> list[list[str]]", "docstring": "Extract normalized string rows from a worksheet.\n\nArgs:\n sheet (int | str | None):\n Worksheet index or title; defaults to the first worksheet.\n skip_empty (bool):\n When True (default), rows whose cells are all blank are omitted.\n\nReturns:\n list[list[str]]:\n Normalized rows; trailing blank cells are trimmed per row.\n\nRaises:\n ValueError:\n If the requested sheet does not exist." } } @@ -2793,74 +2443,53 @@ "signature": null, "docstring": "# Summary\n\nXLSX parser base implementation for OmniRead.\n\nThis module defines the **XLSX-specific parser contract**, extending the\nformat-agnostic `BaseParser` with constraints appropriate for Office Open\nXML spreadsheet content.", "members": { - "abstractmethod": { - "name": "abstractmethod", - "kind": "alias", - "path": "omniread.xlsx.parser_base.abstractmethod", - "signature": "", - "docstring": null - }, - "Generic": { - "name": "Generic", - "kind": "alias", - "path": "omniread.xlsx.parser_base.Generic", - "signature": "", - "docstring": null - }, - "TypeVar": { - "name": "TypeVar", - "kind": "alias", - "path": "omniread.xlsx.parser_base.TypeVar", - "signature": "", - "docstring": null - }, "ContentType": { "name": "ContentType", "kind": "class", "path": "omniread.xlsx.parser_base.ContentType", - "signature": "", + "signature": null, "docstring": "Supported MIME types for extracted content.\n\nNotes:\n **Guarantees:**\n\n - This enum represents the declared or inferred media type of the\n content source.\n - It is primarily used for routing content to the appropriate\n parser or downstream consumer.", "members": { "HTML": { "name": "HTML", "kind": "attribute", "path": "omniread.xlsx.parser_base.ContentType.HTML", - "signature": "", + "signature": null, "docstring": "HTML document content." }, "PDF": { "name": "PDF", "kind": "attribute", "path": "omniread.xlsx.parser_base.ContentType.PDF", - "signature": "", + "signature": null, "docstring": "PDF document content." }, "XLSX": { "name": "XLSX", "kind": "attribute", "path": "omniread.xlsx.parser_base.ContentType.XLSX", - "signature": "", + "signature": null, "docstring": "Office Open XML spreadsheet (xlsx/xlsm) content." }, "CSV": { "name": "CSV", "kind": "attribute", "path": "omniread.xlsx.parser_base.ContentType.CSV", - "signature": "", + "signature": null, "docstring": "Comma-separated-value document content." }, "JSON": { "name": "JSON", "kind": "attribute", "path": "omniread.xlsx.parser_base.ContentType.JSON", - "signature": "", + "signature": null, "docstring": "JSON document content." }, "XML": { "name": "XML", "kind": "attribute", "path": "omniread.xlsx.parser_base.ContentType.XML", - "signature": "", + "signature": null, "docstring": "XML document content." } } @@ -2869,35 +2498,35 @@ "name": "BaseParser", "kind": "class", "path": "omniread.xlsx.parser_base.BaseParser", - "signature": "", + "signature": "BaseParser(content: Content)", "docstring": "Base interface for all parsers.\n\nNotes:\n **Guarantees:**\n\n - A parser is a self-contained object that owns the `Content` it is\n responsible for interpreting.\n - Consumers may rely on early validation of content compatibility\n and type-stable return values from `parse()`.\n\n **Responsibilities:**\n\n - Implementations must declare supported content types via `supported_types`.\n - Implementations must raise parsing-specific exceptions from `parse()`.\n - Implementations must remain deterministic for a given input.", "members": { "supported_types": { "name": "supported_types", "kind": "attribute", "path": "omniread.xlsx.parser_base.BaseParser.supported_types", - "signature": "", + "signature": null, "docstring": "Set of content types supported by this parser. An empty set indicates that the parser is content-type agnostic." }, "content": { "name": "content", "kind": "attribute", "path": "omniread.xlsx.parser_base.BaseParser.content", - "signature": "", + "signature": null, "docstring": null }, "parse": { "name": "parse", "kind": "function", "path": "omniread.xlsx.parser_base.BaseParser.parse", - "signature": "", + "signature": "parse()", "docstring": "Parse the owned content into structured output.\n\nReturns:\n T:\n Parsed, structured representation.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must fully consume the provided content and\n return a deterministic, structured output." }, "supports": { "name": "supports", "kind": "function", "path": "omniread.xlsx.parser_base.BaseParser.supports", - "signature": "", + "signature": "supports()", "docstring": "Check whether this parser supports the content's type.\n\nReturns:\n bool:\n True if the content type is supported; False otherwise." } } @@ -2913,7 +2542,7 @@ "name": "XlsxParserBase", "kind": "class", "path": "omniread.xlsx.parser_base.XlsxParserBase", - "signature": "", + "signature": null, "docstring": "Base xlsx parser.\n\nNotes:\n **Responsibilities:**\n\n - This class enforces xlsx content-type compatibility and provides\n the extension point for implementing concrete xlsx parsing\n strategies.\n\n **Constraints:**\n\n - Concrete implementations must define the output type `T` and\n implement the `parse()` method.", "members": { "supported_types": { @@ -2927,7 +2556,7 @@ "name": "parse", "kind": "function", "path": "omniread.xlsx.parser_base.XlsxParserBase.parse", - "signature": "", + "signature": "parse() -> T", "docstring": "Parse xlsx content into a structured output.\n\nReturns:\n T:\n Parsed representation of type `T`.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation." } } @@ -2941,53 +2570,39 @@ "signature": null, "docstring": "# Summary\n\nXLSX scraper for OmniRead.\n\nThis module defines the scraper responsible for acquiring raw Office Open\nXML spreadsheet content from a backing store via a configured client.\n\nThe scraper does not interpret or parse the acquired bytes; it wraps them in\nthe canonical `Content` model.", "members": { - "Mapping": { - "name": "Mapping", - "kind": "alias", - "path": "omniread.xlsx.scraper.Mapping", - "signature": "", - "docstring": null - }, - "Any": { - "name": "Any", - "kind": "alias", - "path": "omniread.xlsx.scraper.Any", - "signature": "", - "docstring": null - }, "Content": { "name": "Content", "kind": "class", "path": "omniread.xlsx.scraper.Content", - "signature": "", + "signature": "Content(raw: bytes, source: str, content_type: ContentType | None = ..., metadata: Mapping[str, Any] | None = ...)", "docstring": "Normalized representation of extracted content.\n\nNotes:\n **Responsibilities:**\n\n - A `Content` instance represents a raw content payload along with\n minimal contextual metadata describing its origin and type.\n - This class is the primary exchange format between scrapers,\n parsers, and downstream consumers.", "members": { "raw": { "name": "raw", "kind": "attribute", "path": "omniread.xlsx.scraper.Content.raw", - "signature": "", + "signature": null, "docstring": "Raw content bytes as retrieved from the source." }, "source": { "name": "source", "kind": "attribute", "path": "omniread.xlsx.scraper.Content.source", - "signature": "", + "signature": null, "docstring": "Identifier of the content origin (URL, file path, or logical name)." }, "content_type": { "name": "content_type", "kind": "attribute", "path": "omniread.xlsx.scraper.Content.content_type", - "signature": "", + "signature": null, "docstring": "Optional MIME type of the content, if known." }, "metadata": { "name": "metadata", "kind": "attribute", "path": "omniread.xlsx.scraper.Content.metadata", - "signature": "", + "signature": null, "docstring": "Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes)." } } @@ -2996,49 +2611,49 @@ "name": "ContentType", "kind": "class", "path": "omniread.xlsx.scraper.ContentType", - "signature": "", + "signature": null, "docstring": "Supported MIME types for extracted content.\n\nNotes:\n **Guarantees:**\n\n - This enum represents the declared or inferred media type of the\n content source.\n - It is primarily used for routing content to the appropriate\n parser or downstream consumer.", "members": { "HTML": { "name": "HTML", "kind": "attribute", "path": "omniread.xlsx.scraper.ContentType.HTML", - "signature": "", + "signature": null, "docstring": "HTML document content." }, "PDF": { "name": "PDF", "kind": "attribute", "path": "omniread.xlsx.scraper.ContentType.PDF", - "signature": "", + "signature": null, "docstring": "PDF document content." }, "XLSX": { "name": "XLSX", "kind": "attribute", "path": "omniread.xlsx.scraper.ContentType.XLSX", - "signature": "", + "signature": null, "docstring": "Office Open XML spreadsheet (xlsx/xlsm) content." }, "CSV": { "name": "CSV", "kind": "attribute", "path": "omniread.xlsx.scraper.ContentType.CSV", - "signature": "", + "signature": null, "docstring": "Comma-separated-value document content." }, "JSON": { "name": "JSON", "kind": "attribute", "path": "omniread.xlsx.scraper.ContentType.JSON", - "signature": "", + "signature": null, "docstring": "JSON document content." }, "XML": { "name": "XML", "kind": "attribute", "path": "omniread.xlsx.scraper.ContentType.XML", - "signature": "", + "signature": null, "docstring": "XML document content." } } @@ -3047,14 +2662,14 @@ "name": "BaseXlsxClient", "kind": "class", "path": "omniread.xlsx.scraper.BaseXlsxClient", - "signature": "", + "signature": null, "docstring": "Abstract client responsible for retrieving spreadsheet bytes.\n\nRetrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).\n\nNotes:\n **Responsibilities:**\n\n - Implementations must accept a source identifier appropriate to\n the backing store.\n - Return the full xlsx binary payload.\n - Raise retrieval-specific errors on failure.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.xlsx.scraper.BaseXlsxClient.fetch", - "signature": "", + "signature": "fetch(source: Any)", "docstring": "Fetch raw xlsx bytes from the given source.\n\nArgs:\n source (Any):\n Identifier of the spreadsheet location, such as a file path,\n object storage key, or remote reference.\n\nReturns:\n bytes:\n Raw xlsx bytes.\n\nRaises:\n Exception:\n Retrieval-specific errors defined by the implementation." } } @@ -3063,14 +2678,14 @@ "name": "XlsxScraper", "kind": "class", "path": "omniread.xlsx.scraper.XlsxScraper", - "signature": "", + "signature": "XlsxScraper(*, client: BaseXlsxClient)", "docstring": "Scraper for xlsx spreadsheet documents.\n\nNotes:\n **Responsibilities:**\n\n - Fetch raw xlsx bytes via the configured client.\n - Wrap the payload in a canonical `Content` instance with the\n XLSX content type and source identifier.\n\n **Constraints:**\n\n - The scraper does not perform parsing or interpretation.\n - Does not assume a specific storage backend.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.xlsx.scraper.XlsxScraper.fetch", - "signature": "", + "signature": "fetch(source: Any, *, metadata: Mapping[str, Any] | None = None) -> Content", "docstring": "Fetch an xlsx document from the given source.\n\nArgs:\n source (Any):\n Identifier of the spreadsheet source as understood by the\n configured client.\n metadata (Mapping[str, Any] | None, optional):\n Optional metadata to attach to the returned content.\n\nReturns:\n Content:\n A `Content` instance containing raw xlsx bytes, source\n identifier, XLSX content type, and optional metadata.\n\nRaises:\n Exception:\n Retrieval-specific errors raised by the client." } } diff --git a/mcp/omniread/modules/omniread.pdf.client.json b/mcp/omniread/modules/omniread.pdf.client.json index ea9dcfe..b73e277 100644 --- a/mcp/omniread/modules/omniread.pdf.client.json +++ b/mcp/omniread/modules/omniread.pdf.client.json @@ -4,46 +4,18 @@ "path": "omniread.pdf.client", "docstring": "# Summary\n\nPDF client abstractions for OmniRead.\n\nThis module defines the **client layer** responsible for retrieving raw PDF\nbytes from a concrete backing store.\n\nClients provide low-level access to PDF binaries and are intentionally\ndecoupled from scraping and parsing logic. They do not perform validation,\ninterpretation, or content extraction.\n\nTypical backing stores include:\n\n- Local filesystems\n- Object storage (S3, GCS, etc.)\n- Network file systems", "objects": { - "ABC": { - "name": "ABC", - "kind": "alias", - "path": "omniread.pdf.client.ABC", - "signature": "", - "docstring": null - }, - "abstractmethod": { - "name": "abstractmethod", - "kind": "alias", - "path": "omniread.pdf.client.abstractmethod", - "signature": "", - "docstring": null - }, - "Path": { - "name": "Path", - "kind": "alias", - "path": "omniread.pdf.client.Path", - "signature": "", - "docstring": null - }, - "Any": { - "name": "Any", - "kind": "alias", - "path": "omniread.pdf.client.Any", - "signature": "", - "docstring": null - }, "BasePDFClient": { "name": "BasePDFClient", "kind": "class", "path": "omniread.pdf.client.BasePDFClient", - "signature": "", + "signature": null, "docstring": "Abstract client responsible for retrieving PDF bytes.\n\nRetrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).\n\nNotes:\n **Responsibilities:**\n\n - Implementations must accept a source identifier appropriate to\n the backing store.\n - Return the full PDF binary payload.\n - Raise retrieval-specific errors on failure.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.pdf.client.BasePDFClient.fetch", - "signature": "", + "signature": "fetch(source: Any) -> bytes", "docstring": "Fetch raw PDF bytes from the given source.\n\nArgs:\n source (Any):\n Identifier of the PDF location, such as a file path, object storage key, or remote reference.\n\nReturns:\n bytes:\n Raw PDF bytes.\n\nRaises:\n Exception:\n Retrieval-specific errors defined by the implementation." } } @@ -52,14 +24,14 @@ "name": "FileSystemPDFClient", "kind": "class", "path": "omniread.pdf.client.FileSystemPDFClient", - "signature": "", + "signature": null, "docstring": "PDF client that reads from the local filesystem.\n\nNotes:\n **Guarantees:**\n\n - This client reads PDF files directly from the disk and returns\n their raw binary contents.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.pdf.client.FileSystemPDFClient.fetch", - "signature": "", + "signature": "fetch(path: Path) -> bytes", "docstring": "Read a PDF file from the local filesystem.\n\nArgs:\n path (Path):\n Filesystem path to the PDF file.\n\nReturns:\n bytes:\n Raw PDF bytes.\n\nRaises:\n FileNotFoundError:\n If the path does not exist.\n ValueError:\n If the path exists but is not a file." } } diff --git a/mcp/omniread/modules/omniread.pdf.json b/mcp/omniread/modules/omniread.pdf.json index 28d9555..1695b44 100644 --- a/mcp/omniread/modules/omniread.pdf.json +++ b/mcp/omniread/modules/omniread.pdf.json @@ -8,14 +8,14 @@ "name": "FileSystemPDFClient", "kind": "class", "path": "omniread.pdf.FileSystemPDFClient", - "signature": "", + "signature": null, "docstring": "PDF client that reads from the local filesystem.\n\nNotes:\n **Guarantees:**\n\n - This client reads PDF files directly from the disk and returns\n their raw binary contents.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.pdf.FileSystemPDFClient.fetch", - "signature": "", + "signature": "fetch(path: Path)", "docstring": "Read a PDF file from the local filesystem.\n\nArgs:\n path (Path):\n Filesystem path to the PDF file.\n\nReturns:\n bytes:\n Raw PDF bytes.\n\nRaises:\n FileNotFoundError:\n If the path does not exist.\n ValueError:\n If the path exists but is not a file." } } @@ -24,14 +24,14 @@ "name": "PDFScraper", "kind": "class", "path": "omniread.pdf.PDFScraper", - "signature": "", + "signature": "PDFScraper(*, client: BasePDFClient)", "docstring": "Scraper for PDF sources.\n\nNotes:\n **Responsibilities:**\n\n - Delegates byte retrieval to a PDF client and normalizes output\n into `Content`.\n - Preserves caller-provided metadata.\n\n **Constraints:**\n\n - The scraper does not perform parsing or interpretation.\n - Does not assume a specific storage backend.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.pdf.PDFScraper.fetch", - "signature": "", + "signature": "fetch(source: Any, *, metadata: Mapping[str, Any] | None = None)", "docstring": "Fetch a PDF document from the given source.\n\nArgs:\n source (Any):\n Identifier of the PDF source as understood by the configured PDF client.\n metadata (Mapping[str, Any] | None, optional):\n Optional metadata to attach to the returned content.\n\nReturns:\n Content:\n A `Content` instance containing raw PDF bytes, source identifier, PDF content type, and optional metadata.\n\nRaises:\n Exception:\n Retrieval-specific errors raised by the PDF client." } } @@ -40,21 +40,21 @@ "name": "PDFParser", "kind": "class", "path": "omniread.pdf.PDFParser", - "signature": "", + "signature": null, "docstring": "Base PDF parser.\n\nNotes:\n **Responsibilities:**\n\n - This class enforces PDF content-type compatibility and provides\n the extension point for implementing concrete PDF parsing strategies.\n\n **Constraints:**\n\n - Concrete implementations must define the output type `T` and\n implement the `parse()` method.", "members": { "supported_types": { "name": "supported_types", "kind": "attribute", "path": "omniread.pdf.PDFParser.supported_types", - "signature": "", + "signature": null, "docstring": "Set of content types supported by this parser (PDF only)." }, "parse": { "name": "parse", "kind": "function", "path": "omniread.pdf.PDFParser.parse", - "signature": "", + "signature": "parse()", "docstring": "Parse PDF content into a structured output.\n\nReturns:\n T:\n Parsed representation of type `T`.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must fully interpret the PDF binary payload and\n return a deterministic, structured output." } } @@ -66,46 +66,18 @@ "signature": null, "docstring": "# Summary\n\nPDF client abstractions for OmniRead.\n\nThis module defines the **client layer** responsible for retrieving raw PDF\nbytes from a concrete backing store.\n\nClients provide low-level access to PDF binaries and are intentionally\ndecoupled from scraping and parsing logic. They do not perform validation,\ninterpretation, or content extraction.\n\nTypical backing stores include:\n\n- Local filesystems\n- Object storage (S3, GCS, etc.)\n- Network file systems", "members": { - "ABC": { - "name": "ABC", - "kind": "alias", - "path": "omniread.pdf.client.ABC", - "signature": "", - "docstring": null - }, - "abstractmethod": { - "name": "abstractmethod", - "kind": "alias", - "path": "omniread.pdf.client.abstractmethod", - "signature": "", - "docstring": null - }, - "Path": { - "name": "Path", - "kind": "alias", - "path": "omniread.pdf.client.Path", - "signature": "", - "docstring": null - }, - "Any": { - "name": "Any", - "kind": "alias", - "path": "omniread.pdf.client.Any", - "signature": "", - "docstring": null - }, "BasePDFClient": { "name": "BasePDFClient", "kind": "class", "path": "omniread.pdf.client.BasePDFClient", - "signature": "", + "signature": null, "docstring": "Abstract client responsible for retrieving PDF bytes.\n\nRetrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).\n\nNotes:\n **Responsibilities:**\n\n - Implementations must accept a source identifier appropriate to\n the backing store.\n - Return the full PDF binary payload.\n - Raise retrieval-specific errors on failure.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.pdf.client.BasePDFClient.fetch", - "signature": "", + "signature": "fetch(source: Any) -> bytes", "docstring": "Fetch raw PDF bytes from the given source.\n\nArgs:\n source (Any):\n Identifier of the PDF location, such as a file path, object storage key, or remote reference.\n\nReturns:\n bytes:\n Raw PDF bytes.\n\nRaises:\n Exception:\n Retrieval-specific errors defined by the implementation." } } @@ -114,14 +86,14 @@ "name": "FileSystemPDFClient", "kind": "class", "path": "omniread.pdf.client.FileSystemPDFClient", - "signature": "", + "signature": null, "docstring": "PDF client that reads from the local filesystem.\n\nNotes:\n **Guarantees:**\n\n - This client reads PDF files directly from the disk and returns\n their raw binary contents.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.pdf.client.FileSystemPDFClient.fetch", - "signature": "", + "signature": "fetch(path: Path) -> bytes", "docstring": "Read a PDF file from the local filesystem.\n\nArgs:\n path (Path):\n Filesystem path to the PDF file.\n\nReturns:\n bytes:\n Raw PDF bytes.\n\nRaises:\n FileNotFoundError:\n If the path does not exist.\n ValueError:\n If the path exists but is not a file." } } @@ -135,74 +107,53 @@ "signature": null, "docstring": "# Summary\n\nPDF parser base implementations for OmniRead.\n\nThis module defines the **PDF-specific parser contract**, extending the\nformat-agnostic `BaseParser` with constraints appropriate for PDF content.\n\nPDF parsers are responsible for interpreting binary PDF data and producing\nstructured representations suitable for downstream consumption.", "members": { - "abstractmethod": { - "name": "abstractmethod", - "kind": "alias", - "path": "omniread.pdf.parser.abstractmethod", - "signature": "", - "docstring": null - }, - "Generic": { - "name": "Generic", - "kind": "alias", - "path": "omniread.pdf.parser.Generic", - "signature": "", - "docstring": null - }, - "TypeVar": { - "name": "TypeVar", - "kind": "alias", - "path": "omniread.pdf.parser.TypeVar", - "signature": "", - "docstring": null - }, "ContentType": { "name": "ContentType", "kind": "class", "path": "omniread.pdf.parser.ContentType", - "signature": "", + "signature": null, "docstring": "Supported MIME types for extracted content.\n\nNotes:\n **Guarantees:**\n\n - This enum represents the declared or inferred media type of the\n content source.\n - It is primarily used for routing content to the appropriate\n parser or downstream consumer.", "members": { "HTML": { "name": "HTML", "kind": "attribute", "path": "omniread.pdf.parser.ContentType.HTML", - "signature": "", + "signature": null, "docstring": "HTML document content." }, "PDF": { "name": "PDF", "kind": "attribute", "path": "omniread.pdf.parser.ContentType.PDF", - "signature": "", + "signature": null, "docstring": "PDF document content." }, "XLSX": { "name": "XLSX", "kind": "attribute", "path": "omniread.pdf.parser.ContentType.XLSX", - "signature": "", + "signature": null, "docstring": "Office Open XML spreadsheet (xlsx/xlsm) content." }, "CSV": { "name": "CSV", "kind": "attribute", "path": "omniread.pdf.parser.ContentType.CSV", - "signature": "", + "signature": null, "docstring": "Comma-separated-value document content." }, "JSON": { "name": "JSON", "kind": "attribute", "path": "omniread.pdf.parser.ContentType.JSON", - "signature": "", + "signature": null, "docstring": "JSON document content." }, "XML": { "name": "XML", "kind": "attribute", "path": "omniread.pdf.parser.ContentType.XML", - "signature": "", + "signature": null, "docstring": "XML document content." } } @@ -211,35 +162,35 @@ "name": "BaseParser", "kind": "class", "path": "omniread.pdf.parser.BaseParser", - "signature": "", + "signature": "BaseParser(content: Content)", "docstring": "Base interface for all parsers.\n\nNotes:\n **Guarantees:**\n\n - A parser is a self-contained object that owns the `Content` it is\n responsible for interpreting.\n - Consumers may rely on early validation of content compatibility\n and type-stable return values from `parse()`.\n\n **Responsibilities:**\n\n - Implementations must declare supported content types via `supported_types`.\n - Implementations must raise parsing-specific exceptions from `parse()`.\n - Implementations must remain deterministic for a given input.", "members": { "supported_types": { "name": "supported_types", "kind": "attribute", "path": "omniread.pdf.parser.BaseParser.supported_types", - "signature": "", + "signature": null, "docstring": "Set of content types supported by this parser. An empty set indicates that the parser is content-type agnostic." }, "content": { "name": "content", "kind": "attribute", "path": "omniread.pdf.parser.BaseParser.content", - "signature": "", + "signature": null, "docstring": null }, "parse": { "name": "parse", "kind": "function", "path": "omniread.pdf.parser.BaseParser.parse", - "signature": "", + "signature": "parse()", "docstring": "Parse the owned content into structured output.\n\nReturns:\n T:\n Parsed, structured representation.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must fully consume the provided content and\n return a deterministic, structured output." }, "supports": { "name": "supports", "kind": "function", "path": "omniread.pdf.parser.BaseParser.supports", - "signature": "", + "signature": "supports()", "docstring": "Check whether this parser supports the content's type.\n\nReturns:\n bool:\n True if the content type is supported; False otherwise." } } @@ -255,7 +206,7 @@ "name": "PDFParser", "kind": "class", "path": "omniread.pdf.parser.PDFParser", - "signature": "", + "signature": null, "docstring": "Base PDF parser.\n\nNotes:\n **Responsibilities:**\n\n - This class enforces PDF content-type compatibility and provides\n the extension point for implementing concrete PDF parsing strategies.\n\n **Constraints:**\n\n - Concrete implementations must define the output type `T` and\n implement the `parse()` method.", "members": { "supported_types": { @@ -269,7 +220,7 @@ "name": "parse", "kind": "function", "path": "omniread.pdf.parser.PDFParser.parse", - "signature": "", + "signature": "parse() -> T", "docstring": "Parse PDF content into a structured output.\n\nReturns:\n T:\n Parsed representation of type `T`.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must fully interpret the PDF binary payload and\n return a deterministic, structured output." } } @@ -283,53 +234,39 @@ "signature": null, "docstring": "# Summary\n\nPDF scraping implementation for OmniRead.\n\nThis module provides a PDF-specific scraper that coordinates PDF byte\nretrieval via a client and normalizes the result into a `Content` object.\n\nThe scraper implements the core `BaseScraper` contract while delegating\nall storage and access concerns to a `BasePDFClient` implementation.", "members": { - "Mapping": { - "name": "Mapping", - "kind": "alias", - "path": "omniread.pdf.scraper.Mapping", - "signature": "", - "docstring": null - }, - "Any": { - "name": "Any", - "kind": "alias", - "path": "omniread.pdf.scraper.Any", - "signature": "", - "docstring": null - }, "Content": { "name": "Content", "kind": "class", "path": "omniread.pdf.scraper.Content", - "signature": "", + "signature": "Content(raw: bytes, source: str, content_type: ContentType | None = ..., metadata: Mapping[str, Any] | None = ...)", "docstring": "Normalized representation of extracted content.\n\nNotes:\n **Responsibilities:**\n\n - A `Content` instance represents a raw content payload along with\n minimal contextual metadata describing its origin and type.\n - This class is the primary exchange format between scrapers,\n parsers, and downstream consumers.", "members": { "raw": { "name": "raw", "kind": "attribute", "path": "omniread.pdf.scraper.Content.raw", - "signature": "", + "signature": null, "docstring": "Raw content bytes as retrieved from the source." }, "source": { "name": "source", "kind": "attribute", "path": "omniread.pdf.scraper.Content.source", - "signature": "", + "signature": null, "docstring": "Identifier of the content origin (URL, file path, or logical name)." }, "content_type": { "name": "content_type", "kind": "attribute", "path": "omniread.pdf.scraper.Content.content_type", - "signature": "", + "signature": null, "docstring": "Optional MIME type of the content, if known." }, "metadata": { "name": "metadata", "kind": "attribute", "path": "omniread.pdf.scraper.Content.metadata", - "signature": "", + "signature": null, "docstring": "Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes)." } } @@ -338,49 +275,49 @@ "name": "ContentType", "kind": "class", "path": "omniread.pdf.scraper.ContentType", - "signature": "", + "signature": null, "docstring": "Supported MIME types for extracted content.\n\nNotes:\n **Guarantees:**\n\n - This enum represents the declared or inferred media type of the\n content source.\n - It is primarily used for routing content to the appropriate\n parser or downstream consumer.", "members": { "HTML": { "name": "HTML", "kind": "attribute", "path": "omniread.pdf.scraper.ContentType.HTML", - "signature": "", + "signature": null, "docstring": "HTML document content." }, "PDF": { "name": "PDF", "kind": "attribute", "path": "omniread.pdf.scraper.ContentType.PDF", - "signature": "", + "signature": null, "docstring": "PDF document content." }, "XLSX": { "name": "XLSX", "kind": "attribute", "path": "omniread.pdf.scraper.ContentType.XLSX", - "signature": "", + "signature": null, "docstring": "Office Open XML spreadsheet (xlsx/xlsm) content." }, "CSV": { "name": "CSV", "kind": "attribute", "path": "omniread.pdf.scraper.ContentType.CSV", - "signature": "", + "signature": null, "docstring": "Comma-separated-value document content." }, "JSON": { "name": "JSON", "kind": "attribute", "path": "omniread.pdf.scraper.ContentType.JSON", - "signature": "", + "signature": null, "docstring": "JSON document content." }, "XML": { "name": "XML", "kind": "attribute", "path": "omniread.pdf.scraper.ContentType.XML", - "signature": "", + "signature": null, "docstring": "XML document content." } } @@ -389,14 +326,14 @@ "name": "BaseScraper", "kind": "class", "path": "omniread.pdf.scraper.BaseScraper", - "signature": "", + "signature": null, "docstring": "Base interface for all scrapers.\n\nNotes:\n **Responsibilities:**\n\n - A scraper is responsible ONLY for fetching raw content (bytes)\n from a source. It must not interpret or parse it.\n - A scraper is a stateless acquisition component that retrieves raw\n content from a source and returns it as a `Content` object.\n - Scrapers define how content is obtained, not what the content means.\n - Implementations may vary in transport mechanism, authentication\n strategy, retry and backoff behavior.\n\n **Constraints:**\n\n - Implementations must not parse content, modify content semantics,\n or couple scraping logic to a specific parser.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.pdf.scraper.BaseScraper.fetch", - "signature": "", + "signature": "fetch(source: str, *, metadata: Mapping[str, Any] | None = None)", "docstring": "Fetch raw content from the given source.\n\nArgs:\n source (str):\n Location identifier (URL, file path, S3 URI, etc.).\n\n metadata (Mapping[str, Any] | None, optional):\n Optional hints for the scraper (headers, auth, etc.).\n\nReturns:\n Content:\n Content object containing raw bytes and metadata.\n\nRaises:\n Exception:\n Retrieval-specific errors as defined by the implementation.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must retrieve the content referenced by `source`\n and return it as raw bytes wrapped in a `Content` object." } } @@ -405,14 +342,14 @@ "name": "BasePDFClient", "kind": "class", "path": "omniread.pdf.scraper.BasePDFClient", - "signature": "", + "signature": null, "docstring": "Abstract client responsible for retrieving PDF bytes.\n\nRetrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).\n\nNotes:\n **Responsibilities:**\n\n - Implementations must accept a source identifier appropriate to\n the backing store.\n - Return the full PDF binary payload.\n - Raise retrieval-specific errors on failure.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.pdf.scraper.BasePDFClient.fetch", - "signature": "", + "signature": "fetch(source: Any)", "docstring": "Fetch raw PDF bytes from the given source.\n\nArgs:\n source (Any):\n Identifier of the PDF location, such as a file path, object storage key, or remote reference.\n\nReturns:\n bytes:\n Raw PDF bytes.\n\nRaises:\n Exception:\n Retrieval-specific errors defined by the implementation." } } @@ -421,14 +358,14 @@ "name": "PDFScraper", "kind": "class", "path": "omniread.pdf.scraper.PDFScraper", - "signature": "", + "signature": "PDFScraper(*, client: BasePDFClient)", "docstring": "Scraper for PDF sources.\n\nNotes:\n **Responsibilities:**\n\n - Delegates byte retrieval to a PDF client and normalizes output\n into `Content`.\n - Preserves caller-provided metadata.\n\n **Constraints:**\n\n - The scraper does not perform parsing or interpretation.\n - Does not assume a specific storage backend.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.pdf.scraper.PDFScraper.fetch", - "signature": "", + "signature": "fetch(source: Any, *, metadata: Mapping[str, Any] | None = None) -> Content", "docstring": "Fetch a PDF document from the given source.\n\nArgs:\n source (Any):\n Identifier of the PDF source as understood by the configured PDF client.\n metadata (Mapping[str, Any] | None, optional):\n Optional metadata to attach to the returned content.\n\nReturns:\n Content:\n A `Content` instance containing raw PDF bytes, source identifier, PDF content type, and optional metadata.\n\nRaises:\n Exception:\n Retrieval-specific errors raised by the PDF client." } } diff --git a/mcp/omniread/modules/omniread.pdf.parser.json b/mcp/omniread/modules/omniread.pdf.parser.json index 8fe1584..e330994 100644 --- a/mcp/omniread/modules/omniread.pdf.parser.json +++ b/mcp/omniread/modules/omniread.pdf.parser.json @@ -4,74 +4,53 @@ "path": "omniread.pdf.parser", "docstring": "# Summary\n\nPDF parser base implementations for OmniRead.\n\nThis module defines the **PDF-specific parser contract**, extending the\nformat-agnostic `BaseParser` with constraints appropriate for PDF content.\n\nPDF parsers are responsible for interpreting binary PDF data and producing\nstructured representations suitable for downstream consumption.", "objects": { - "abstractmethod": { - "name": "abstractmethod", - "kind": "alias", - "path": "omniread.pdf.parser.abstractmethod", - "signature": "", - "docstring": null - }, - "Generic": { - "name": "Generic", - "kind": "alias", - "path": "omniread.pdf.parser.Generic", - "signature": "", - "docstring": null - }, - "TypeVar": { - "name": "TypeVar", - "kind": "alias", - "path": "omniread.pdf.parser.TypeVar", - "signature": "", - "docstring": null - }, "ContentType": { "name": "ContentType", "kind": "class", "path": "omniread.pdf.parser.ContentType", - "signature": "", + "signature": null, "docstring": "Supported MIME types for extracted content.\n\nNotes:\n **Guarantees:**\n\n - This enum represents the declared or inferred media type of the\n content source.\n - It is primarily used for routing content to the appropriate\n parser or downstream consumer.", "members": { "HTML": { "name": "HTML", "kind": "attribute", "path": "omniread.pdf.parser.ContentType.HTML", - "signature": "", + "signature": null, "docstring": "HTML document content." }, "PDF": { "name": "PDF", "kind": "attribute", "path": "omniread.pdf.parser.ContentType.PDF", - "signature": "", + "signature": null, "docstring": "PDF document content." }, "XLSX": { "name": "XLSX", "kind": "attribute", "path": "omniread.pdf.parser.ContentType.XLSX", - "signature": "", + "signature": null, "docstring": "Office Open XML spreadsheet (xlsx/xlsm) content." }, "CSV": { "name": "CSV", "kind": "attribute", "path": "omniread.pdf.parser.ContentType.CSV", - "signature": "", + "signature": null, "docstring": "Comma-separated-value document content." }, "JSON": { "name": "JSON", "kind": "attribute", "path": "omniread.pdf.parser.ContentType.JSON", - "signature": "", + "signature": null, "docstring": "JSON document content." }, "XML": { "name": "XML", "kind": "attribute", "path": "omniread.pdf.parser.ContentType.XML", - "signature": "", + "signature": null, "docstring": "XML document content." } } @@ -80,35 +59,35 @@ "name": "BaseParser", "kind": "class", "path": "omniread.pdf.parser.BaseParser", - "signature": "", + "signature": "BaseParser(content: Content)", "docstring": "Base interface for all parsers.\n\nNotes:\n **Guarantees:**\n\n - A parser is a self-contained object that owns the `Content` it is\n responsible for interpreting.\n - Consumers may rely on early validation of content compatibility\n and type-stable return values from `parse()`.\n\n **Responsibilities:**\n\n - Implementations must declare supported content types via `supported_types`.\n - Implementations must raise parsing-specific exceptions from `parse()`.\n - Implementations must remain deterministic for a given input.", "members": { "supported_types": { "name": "supported_types", "kind": "attribute", "path": "omniread.pdf.parser.BaseParser.supported_types", - "signature": "", + "signature": null, "docstring": "Set of content types supported by this parser. An empty set indicates that the parser is content-type agnostic." }, "content": { "name": "content", "kind": "attribute", "path": "omniread.pdf.parser.BaseParser.content", - "signature": "", + "signature": null, "docstring": null }, "parse": { "name": "parse", "kind": "function", "path": "omniread.pdf.parser.BaseParser.parse", - "signature": "", + "signature": "parse()", "docstring": "Parse the owned content into structured output.\n\nReturns:\n T:\n Parsed, structured representation.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must fully consume the provided content and\n return a deterministic, structured output." }, "supports": { "name": "supports", "kind": "function", "path": "omniread.pdf.parser.BaseParser.supports", - "signature": "", + "signature": "supports()", "docstring": "Check whether this parser supports the content's type.\n\nReturns:\n bool:\n True if the content type is supported; False otherwise." } } @@ -124,7 +103,7 @@ "name": "PDFParser", "kind": "class", "path": "omniread.pdf.parser.PDFParser", - "signature": "", + "signature": null, "docstring": "Base PDF parser.\n\nNotes:\n **Responsibilities:**\n\n - This class enforces PDF content-type compatibility and provides\n the extension point for implementing concrete PDF parsing strategies.\n\n **Constraints:**\n\n - Concrete implementations must define the output type `T` and\n implement the `parse()` method.", "members": { "supported_types": { @@ -138,7 +117,7 @@ "name": "parse", "kind": "function", "path": "omniread.pdf.parser.PDFParser.parse", - "signature": "", + "signature": "parse() -> T", "docstring": "Parse PDF content into a structured output.\n\nReturns:\n T:\n Parsed representation of type `T`.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must fully interpret the PDF binary payload and\n return a deterministic, structured output." } } diff --git a/mcp/omniread/modules/omniread.pdf.scraper.json b/mcp/omniread/modules/omniread.pdf.scraper.json index 7f07064..ff03d34 100644 --- a/mcp/omniread/modules/omniread.pdf.scraper.json +++ b/mcp/omniread/modules/omniread.pdf.scraper.json @@ -4,53 +4,39 @@ "path": "omniread.pdf.scraper", "docstring": "# Summary\n\nPDF scraping implementation for OmniRead.\n\nThis module provides a PDF-specific scraper that coordinates PDF byte\nretrieval via a client and normalizes the result into a `Content` object.\n\nThe scraper implements the core `BaseScraper` contract while delegating\nall storage and access concerns to a `BasePDFClient` implementation.", "objects": { - "Mapping": { - "name": "Mapping", - "kind": "alias", - "path": "omniread.pdf.scraper.Mapping", - "signature": "", - "docstring": null - }, - "Any": { - "name": "Any", - "kind": "alias", - "path": "omniread.pdf.scraper.Any", - "signature": "", - "docstring": null - }, "Content": { "name": "Content", "kind": "class", "path": "omniread.pdf.scraper.Content", - "signature": "", + "signature": "Content(raw: bytes, source: str, content_type: ContentType | None = ..., metadata: Mapping[str, Any] | None = ...)", "docstring": "Normalized representation of extracted content.\n\nNotes:\n **Responsibilities:**\n\n - A `Content` instance represents a raw content payload along with\n minimal contextual metadata describing its origin and type.\n - This class is the primary exchange format between scrapers,\n parsers, and downstream consumers.", "members": { "raw": { "name": "raw", "kind": "attribute", "path": "omniread.pdf.scraper.Content.raw", - "signature": "", + "signature": null, "docstring": "Raw content bytes as retrieved from the source." }, "source": { "name": "source", "kind": "attribute", "path": "omniread.pdf.scraper.Content.source", - "signature": "", + "signature": null, "docstring": "Identifier of the content origin (URL, file path, or logical name)." }, "content_type": { "name": "content_type", "kind": "attribute", "path": "omniread.pdf.scraper.Content.content_type", - "signature": "", + "signature": null, "docstring": "Optional MIME type of the content, if known." }, "metadata": { "name": "metadata", "kind": "attribute", "path": "omniread.pdf.scraper.Content.metadata", - "signature": "", + "signature": null, "docstring": "Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes)." } } @@ -59,49 +45,49 @@ "name": "ContentType", "kind": "class", "path": "omniread.pdf.scraper.ContentType", - "signature": "", + "signature": null, "docstring": "Supported MIME types for extracted content.\n\nNotes:\n **Guarantees:**\n\n - This enum represents the declared or inferred media type of the\n content source.\n - It is primarily used for routing content to the appropriate\n parser or downstream consumer.", "members": { "HTML": { "name": "HTML", "kind": "attribute", "path": "omniread.pdf.scraper.ContentType.HTML", - "signature": "", + "signature": null, "docstring": "HTML document content." }, "PDF": { "name": "PDF", "kind": "attribute", "path": "omniread.pdf.scraper.ContentType.PDF", - "signature": "", + "signature": null, "docstring": "PDF document content." }, "XLSX": { "name": "XLSX", "kind": "attribute", "path": "omniread.pdf.scraper.ContentType.XLSX", - "signature": "", + "signature": null, "docstring": "Office Open XML spreadsheet (xlsx/xlsm) content." }, "CSV": { "name": "CSV", "kind": "attribute", "path": "omniread.pdf.scraper.ContentType.CSV", - "signature": "", + "signature": null, "docstring": "Comma-separated-value document content." }, "JSON": { "name": "JSON", "kind": "attribute", "path": "omniread.pdf.scraper.ContentType.JSON", - "signature": "", + "signature": null, "docstring": "JSON document content." }, "XML": { "name": "XML", "kind": "attribute", "path": "omniread.pdf.scraper.ContentType.XML", - "signature": "", + "signature": null, "docstring": "XML document content." } } @@ -110,14 +96,14 @@ "name": "BaseScraper", "kind": "class", "path": "omniread.pdf.scraper.BaseScraper", - "signature": "", + "signature": null, "docstring": "Base interface for all scrapers.\n\nNotes:\n **Responsibilities:**\n\n - A scraper is responsible ONLY for fetching raw content (bytes)\n from a source. It must not interpret or parse it.\n - A scraper is a stateless acquisition component that retrieves raw\n content from a source and returns it as a `Content` object.\n - Scrapers define how content is obtained, not what the content means.\n - Implementations may vary in transport mechanism, authentication\n strategy, retry and backoff behavior.\n\n **Constraints:**\n\n - Implementations must not parse content, modify content semantics,\n or couple scraping logic to a specific parser.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.pdf.scraper.BaseScraper.fetch", - "signature": "", + "signature": "fetch(source: str, *, metadata: Mapping[str, Any] | None = None)", "docstring": "Fetch raw content from the given source.\n\nArgs:\n source (str):\n Location identifier (URL, file path, S3 URI, etc.).\n\n metadata (Mapping[str, Any] | None, optional):\n Optional hints for the scraper (headers, auth, etc.).\n\nReturns:\n Content:\n Content object containing raw bytes and metadata.\n\nRaises:\n Exception:\n Retrieval-specific errors as defined by the implementation.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must retrieve the content referenced by `source`\n and return it as raw bytes wrapped in a `Content` object." } } @@ -126,14 +112,14 @@ "name": "BasePDFClient", "kind": "class", "path": "omniread.pdf.scraper.BasePDFClient", - "signature": "", + "signature": null, "docstring": "Abstract client responsible for retrieving PDF bytes.\n\nRetrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).\n\nNotes:\n **Responsibilities:**\n\n - Implementations must accept a source identifier appropriate to\n the backing store.\n - Return the full PDF binary payload.\n - Raise retrieval-specific errors on failure.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.pdf.scraper.BasePDFClient.fetch", - "signature": "", + "signature": "fetch(source: Any)", "docstring": "Fetch raw PDF bytes from the given source.\n\nArgs:\n source (Any):\n Identifier of the PDF location, such as a file path, object storage key, or remote reference.\n\nReturns:\n bytes:\n Raw PDF bytes.\n\nRaises:\n Exception:\n Retrieval-specific errors defined by the implementation." } } @@ -142,14 +128,14 @@ "name": "PDFScraper", "kind": "class", "path": "omniread.pdf.scraper.PDFScraper", - "signature": "", + "signature": "PDFScraper(*, client: BasePDFClient)", "docstring": "Scraper for PDF sources.\n\nNotes:\n **Responsibilities:**\n\n - Delegates byte retrieval to a PDF client and normalizes output\n into `Content`.\n - Preserves caller-provided metadata.\n\n **Constraints:**\n\n - The scraper does not perform parsing or interpretation.\n - Does not assume a specific storage backend.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.pdf.scraper.PDFScraper.fetch", - "signature": "", + "signature": "fetch(source: Any, *, metadata: Mapping[str, Any] | None = None) -> Content", "docstring": "Fetch a PDF document from the given source.\n\nArgs:\n source (Any):\n Identifier of the PDF source as understood by the configured PDF client.\n metadata (Mapping[str, Any] | None, optional):\n Optional metadata to attach to the returned content.\n\nReturns:\n Content:\n A `Content` instance containing raw PDF bytes, source identifier, PDF content type, and optional metadata.\n\nRaises:\n Exception:\n Retrieval-specific errors raised by the PDF client." } } diff --git a/mcp/omniread/modules/omniread.xlsx.client.json b/mcp/omniread/modules/omniread.xlsx.client.json index de78b9a..0234f37 100644 --- a/mcp/omniread/modules/omniread.xlsx.client.json +++ b/mcp/omniread/modules/omniread.xlsx.client.json @@ -4,46 +4,18 @@ "path": "omniread.xlsx.client", "docstring": "# Summary\n\nXLSX client abstractions for OmniRead.\n\nThis module defines the **client layer** responsible for retrieving raw\nOffice Open XML spreadsheet bytes from a concrete backing store.\n\nClients provide low-level access to xlsx binaries and are intentionally\ndecoupled from scraping and parsing logic. They do not perform validation,\ninterpretation, or content extraction.\n\nTypical backing stores include:\n\n- Local filesystems\n- Object storage (S3, GCS, etc.)\n- Network file systems", "objects": { - "ABC": { - "name": "ABC", - "kind": "alias", - "path": "omniread.xlsx.client.ABC", - "signature": "", - "docstring": null - }, - "abstractmethod": { - "name": "abstractmethod", - "kind": "alias", - "path": "omniread.xlsx.client.abstractmethod", - "signature": "", - "docstring": null - }, - "Path": { - "name": "Path", - "kind": "alias", - "path": "omniread.xlsx.client.Path", - "signature": "", - "docstring": null - }, - "Any": { - "name": "Any", - "kind": "alias", - "path": "omniread.xlsx.client.Any", - "signature": "", - "docstring": null - }, "BaseXlsxClient": { "name": "BaseXlsxClient", "kind": "class", "path": "omniread.xlsx.client.BaseXlsxClient", - "signature": "", + "signature": null, "docstring": "Abstract client responsible for retrieving spreadsheet bytes.\n\nRetrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).\n\nNotes:\n **Responsibilities:**\n\n - Implementations must accept a source identifier appropriate to\n the backing store.\n - Return the full xlsx binary payload.\n - Raise retrieval-specific errors on failure.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.xlsx.client.BaseXlsxClient.fetch", - "signature": "", + "signature": "fetch(source: Any) -> bytes", "docstring": "Fetch raw xlsx bytes from the given source.\n\nArgs:\n source (Any):\n Identifier of the spreadsheet location, such as a file path,\n object storage key, or remote reference.\n\nReturns:\n bytes:\n Raw xlsx bytes.\n\nRaises:\n Exception:\n Retrieval-specific errors defined by the implementation." } } @@ -52,14 +24,14 @@ "name": "FileSystemXlsxClient", "kind": "class", "path": "omniread.xlsx.client.FileSystemXlsxClient", - "signature": "", + "signature": null, "docstring": "XLSX client that reads from the local filesystem.\n\nNotes:\n **Guarantees:**\n\n - This client reads spreadsheet files directly from the disk and\n returns their raw binary contents.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.xlsx.client.FileSystemXlsxClient.fetch", - "signature": "", + "signature": "fetch(path: Path) -> bytes", "docstring": "Read an xlsx file from the local filesystem.\n\nArgs:\n path (Path):\n Filesystem path to the spreadsheet file.\n\nReturns:\n bytes:\n Raw xlsx bytes.\n\nRaises:\n FileNotFoundError:\n If the path does not exist.\n ValueError:\n If the path exists but is not a file." } } diff --git a/mcp/omniread/modules/omniread.xlsx.json b/mcp/omniread/modules/omniread.xlsx.json index f939e62..ef391a2 100644 --- a/mcp/omniread/modules/omniread.xlsx.json +++ b/mcp/omniread/modules/omniread.xlsx.json @@ -8,14 +8,14 @@ "name": "BaseXlsxClient", "kind": "class", "path": "omniread.xlsx.BaseXlsxClient", - "signature": "", + "signature": null, "docstring": "Abstract client responsible for retrieving spreadsheet bytes.\n\nRetrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).\n\nNotes:\n **Responsibilities:**\n\n - Implementations must accept a source identifier appropriate to\n the backing store.\n - Return the full xlsx binary payload.\n - Raise retrieval-specific errors on failure.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.xlsx.BaseXlsxClient.fetch", - "signature": "", + "signature": "fetch(source: Any)", "docstring": "Fetch raw xlsx bytes from the given source.\n\nArgs:\n source (Any):\n Identifier of the spreadsheet location, such as a file path,\n object storage key, or remote reference.\n\nReturns:\n bytes:\n Raw xlsx bytes.\n\nRaises:\n Exception:\n Retrieval-specific errors defined by the implementation." } } @@ -24,14 +24,14 @@ "name": "FileSystemXlsxClient", "kind": "class", "path": "omniread.xlsx.FileSystemXlsxClient", - "signature": "", + "signature": null, "docstring": "XLSX client that reads from the local filesystem.\n\nNotes:\n **Guarantees:**\n\n - This client reads spreadsheet files directly from the disk and\n returns their raw binary contents.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.xlsx.FileSystemXlsxClient.fetch", - "signature": "", + "signature": "fetch(path: Path)", "docstring": "Read an xlsx file from the local filesystem.\n\nArgs:\n path (Path):\n Filesystem path to the spreadsheet file.\n\nReturns:\n bytes:\n Raw xlsx bytes.\n\nRaises:\n FileNotFoundError:\n If the path does not exist.\n ValueError:\n If the path exists but is not a file." } } @@ -40,35 +40,35 @@ "name": "XlsxParser", "kind": "class", "path": "omniread.xlsx.XlsxParser", - "signature": "", + "signature": "XlsxParser(content: Content, *, data_only: bool = True, read_only: bool = True)", "docstring": "Generic xlsx parser producing string rows from a worksheet.\n\nNotes:\n **Responsibilities:**\n\n - Lazily load the workbook owned by the parser's content.\n - Normalize cells (including dates and numeric values) into\n deterministic string representations.\n - Expose sheet discovery and row extraction helpers.\n\n **Constraints:**\n\n - Cells are rendered with ``str(value)`` after trimming; date and\n datetime values are rendered in ISO format. Consumers requiring\n locale-specific formatting must convert on their side.", "members": { "workbook": { "name": "workbook", "kind": "attribute", "path": "omniread.xlsx.XlsxParser.workbook", - "signature": "", + "signature": null, "docstring": "The lazily loaded workbook backing this parser's content." }, "sheet_names": { "name": "sheet_names", "kind": "attribute", "path": "omniread.xlsx.XlsxParser.sheet_names", - "signature": "", + "signature": null, "docstring": "Names of all worksheets contained in the workbook." }, "parse": { "name": "parse", "kind": "function", "path": "omniread.xlsx.XlsxParser.parse", - "signature": "", + "signature": "parse()", "docstring": "Parse the first worksheet into normalized string rows.\n\nReturns:\n list[list[str]]:\n Rows of the default (first) worksheet." }, "rows": { "name": "rows", "kind": "function", "path": "omniread.xlsx.XlsxParser.rows", - "signature": "", + "signature": "rows(sheet: int | str | None = None, *, skip_empty: bool = True)", "docstring": "Extract normalized string rows from a worksheet.\n\nArgs:\n sheet (int | str | None):\n Worksheet index or title; defaults to the first worksheet.\n skip_empty (bool):\n When True (default), rows whose cells are all blank are omitted.\n\nReturns:\n list[list[str]]:\n Normalized rows; trailing blank cells are trimmed per row.\n\nRaises:\n ValueError:\n If the requested sheet does not exist." } } @@ -77,21 +77,21 @@ "name": "XlsxParserBase", "kind": "class", "path": "omniread.xlsx.XlsxParserBase", - "signature": "", + "signature": null, "docstring": "Base xlsx parser.\n\nNotes:\n **Responsibilities:**\n\n - This class enforces xlsx content-type compatibility and provides\n the extension point for implementing concrete xlsx parsing\n strategies.\n\n **Constraints:**\n\n - Concrete implementations must define the output type `T` and\n implement the `parse()` method.", "members": { "supported_types": { "name": "supported_types", "kind": "attribute", "path": "omniread.xlsx.XlsxParserBase.supported_types", - "signature": "", + "signature": null, "docstring": "Set of content types supported by this parser (XLSX only)." }, "parse": { "name": "parse", "kind": "function", "path": "omniread.xlsx.XlsxParserBase.parse", - "signature": "", + "signature": "parse()", "docstring": "Parse xlsx content into a structured output.\n\nReturns:\n T:\n Parsed representation of type `T`.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation." } } @@ -100,14 +100,14 @@ "name": "XlsxScraper", "kind": "class", "path": "omniread.xlsx.XlsxScraper", - "signature": "", + "signature": "XlsxScraper(*, client: BaseXlsxClient)", "docstring": "Scraper for xlsx spreadsheet documents.\n\nNotes:\n **Responsibilities:**\n\n - Fetch raw xlsx bytes via the configured client.\n - Wrap the payload in a canonical `Content` instance with the\n XLSX content type and source identifier.\n\n **Constraints:**\n\n - The scraper does not perform parsing or interpretation.\n - Does not assume a specific storage backend.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.xlsx.XlsxScraper.fetch", - "signature": "", + "signature": "fetch(source: Any, *, metadata: Mapping[str, Any] | None = None)", "docstring": "Fetch an xlsx document from the given source.\n\nArgs:\n source (Any):\n Identifier of the spreadsheet source as understood by the\n configured client.\n metadata (Mapping[str, Any] | None, optional):\n Optional metadata to attach to the returned content.\n\nReturns:\n Content:\n A `Content` instance containing raw xlsx bytes, source\n identifier, XLSX content type, and optional metadata.\n\nRaises:\n Exception:\n Retrieval-specific errors raised by the client." } } @@ -119,46 +119,18 @@ "signature": null, "docstring": "# Summary\n\nXLSX client abstractions for OmniRead.\n\nThis module defines the **client layer** responsible for retrieving raw\nOffice Open XML spreadsheet bytes from a concrete backing store.\n\nClients provide low-level access to xlsx binaries and are intentionally\ndecoupled from scraping and parsing logic. They do not perform validation,\ninterpretation, or content extraction.\n\nTypical backing stores include:\n\n- Local filesystems\n- Object storage (S3, GCS, etc.)\n- Network file systems", "members": { - "ABC": { - "name": "ABC", - "kind": "alias", - "path": "omniread.xlsx.client.ABC", - "signature": "", - "docstring": null - }, - "abstractmethod": { - "name": "abstractmethod", - "kind": "alias", - "path": "omniread.xlsx.client.abstractmethod", - "signature": "", - "docstring": null - }, - "Path": { - "name": "Path", - "kind": "alias", - "path": "omniread.xlsx.client.Path", - "signature": "", - "docstring": null - }, - "Any": { - "name": "Any", - "kind": "alias", - "path": "omniread.xlsx.client.Any", - "signature": "", - "docstring": null - }, "BaseXlsxClient": { "name": "BaseXlsxClient", "kind": "class", "path": "omniread.xlsx.client.BaseXlsxClient", - "signature": "", + "signature": null, "docstring": "Abstract client responsible for retrieving spreadsheet bytes.\n\nRetrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).\n\nNotes:\n **Responsibilities:**\n\n - Implementations must accept a source identifier appropriate to\n the backing store.\n - Return the full xlsx binary payload.\n - Raise retrieval-specific errors on failure.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.xlsx.client.BaseXlsxClient.fetch", - "signature": "", + "signature": "fetch(source: Any) -> bytes", "docstring": "Fetch raw xlsx bytes from the given source.\n\nArgs:\n source (Any):\n Identifier of the spreadsheet location, such as a file path,\n object storage key, or remote reference.\n\nReturns:\n bytes:\n Raw xlsx bytes.\n\nRaises:\n Exception:\n Retrieval-specific errors defined by the implementation." } } @@ -167,14 +139,14 @@ "name": "FileSystemXlsxClient", "kind": "class", "path": "omniread.xlsx.client.FileSystemXlsxClient", - "signature": "", + "signature": null, "docstring": "XLSX client that reads from the local filesystem.\n\nNotes:\n **Guarantees:**\n\n - This client reads spreadsheet files directly from the disk and\n returns their raw binary contents.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.xlsx.client.FileSystemXlsxClient.fetch", - "signature": "", + "signature": "fetch(path: Path) -> bytes", "docstring": "Read an xlsx file from the local filesystem.\n\nArgs:\n path (Path):\n Filesystem path to the spreadsheet file.\n\nReturns:\n bytes:\n Raw xlsx bytes.\n\nRaises:\n FileNotFoundError:\n If the path does not exist.\n ValueError:\n If the path exists but is not a file." } } @@ -188,67 +160,39 @@ "signature": null, "docstring": "# Summary\n\nXLSX parser implementations for OmniRead.\n\nThis module provides a concrete, generic parser for Office Open XML\nspreadsheets. It exposes workbook sheets as lists of string rows so\ndownstream consumers can interpret tabular content without depending on\nopenpyxl directly.\n\nThe parser is intentionally statement-agnostic: it performs no header\ndetection or column interpretation beyond basic cell normalization.", "members": { - "datetime": { - "name": "datetime", - "kind": "alias", - "path": "omniread.xlsx.parser.datetime", - "signature": "", - "docstring": null - }, - "BytesIO": { - "name": "BytesIO", - "kind": "alias", - "path": "omniread.xlsx.parser.BytesIO", - "signature": "", - "docstring": null - }, - "Any": { - "name": "Any", - "kind": "alias", - "path": "omniread.xlsx.parser.Any", - "signature": "", - "docstring": null - }, - "openpyxl": { - "name": "openpyxl", - "kind": "alias", - "path": "omniread.xlsx.parser.openpyxl", - "signature": "", - "docstring": null - }, "Content": { "name": "Content", "kind": "class", "path": "omniread.xlsx.parser.Content", - "signature": "", + "signature": "Content(raw: bytes, source: str, content_type: ContentType | None = ..., metadata: Mapping[str, Any] | None = ...)", "docstring": "Normalized representation of extracted content.\n\nNotes:\n **Responsibilities:**\n\n - A `Content` instance represents a raw content payload along with\n minimal contextual metadata describing its origin and type.\n - This class is the primary exchange format between scrapers,\n parsers, and downstream consumers.", "members": { "raw": { "name": "raw", "kind": "attribute", "path": "omniread.xlsx.parser.Content.raw", - "signature": "", + "signature": null, "docstring": "Raw content bytes as retrieved from the source." }, "source": { "name": "source", "kind": "attribute", "path": "omniread.xlsx.parser.Content.source", - "signature": "", + "signature": null, "docstring": "Identifier of the content origin (URL, file path, or logical name)." }, "content_type": { "name": "content_type", "kind": "attribute", "path": "omniread.xlsx.parser.Content.content_type", - "signature": "", + "signature": null, "docstring": "Optional MIME type of the content, if known." }, "metadata": { "name": "metadata", "kind": "attribute", "path": "omniread.xlsx.parser.Content.metadata", - "signature": "", + "signature": null, "docstring": "Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes)." } } @@ -257,21 +201,21 @@ "name": "XlsxParserBase", "kind": "class", "path": "omniread.xlsx.parser.XlsxParserBase", - "signature": "", + "signature": null, "docstring": "Base xlsx parser.\n\nNotes:\n **Responsibilities:**\n\n - This class enforces xlsx content-type compatibility and provides\n the extension point for implementing concrete xlsx parsing\n strategies.\n\n **Constraints:**\n\n - Concrete implementations must define the output type `T` and\n implement the `parse()` method.", "members": { "supported_types": { "name": "supported_types", "kind": "attribute", "path": "omniread.xlsx.parser.XlsxParserBase.supported_types", - "signature": "", + "signature": null, "docstring": "Set of content types supported by this parser (XLSX only)." }, "parse": { "name": "parse", "kind": "function", "path": "omniread.xlsx.parser.XlsxParserBase.parse", - "signature": "", + "signature": "parse()", "docstring": "Parse xlsx content into a structured output.\n\nReturns:\n T:\n Parsed representation of type `T`.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation." } } @@ -280,7 +224,7 @@ "name": "XlsxParser", "kind": "class", "path": "omniread.xlsx.parser.XlsxParser", - "signature": "", + "signature": "XlsxParser(content: Content, *, data_only: bool = True, read_only: bool = True)", "docstring": "Generic xlsx parser producing string rows from a worksheet.\n\nNotes:\n **Responsibilities:**\n\n - Lazily load the workbook owned by the parser's content.\n - Normalize cells (including dates and numeric values) into\n deterministic string representations.\n - Expose sheet discovery and row extraction helpers.\n\n **Constraints:**\n\n - Cells are rendered with ``str(value)`` after trimming; date and\n datetime values are rendered in ISO format. Consumers requiring\n locale-specific formatting must convert on their side.", "members": { "workbook": { @@ -301,14 +245,14 @@ "name": "parse", "kind": "function", "path": "omniread.xlsx.parser.XlsxParser.parse", - "signature": "", + "signature": "parse() -> list[list[str]]", "docstring": "Parse the first worksheet into normalized string rows.\n\nReturns:\n list[list[str]]:\n Rows of the default (first) worksheet." }, "rows": { "name": "rows", "kind": "function", "path": "omniread.xlsx.parser.XlsxParser.rows", - "signature": "", + "signature": "rows(sheet: int | str | None = None, *, skip_empty: bool = True) -> list[list[str]]", "docstring": "Extract normalized string rows from a worksheet.\n\nArgs:\n sheet (int | str | None):\n Worksheet index or title; defaults to the first worksheet.\n skip_empty (bool):\n When True (default), rows whose cells are all blank are omitted.\n\nReturns:\n list[list[str]]:\n Normalized rows; trailing blank cells are trimmed per row.\n\nRaises:\n ValueError:\n If the requested sheet does not exist." } } @@ -322,74 +266,53 @@ "signature": null, "docstring": "# Summary\n\nXLSX parser base implementation for OmniRead.\n\nThis module defines the **XLSX-specific parser contract**, extending the\nformat-agnostic `BaseParser` with constraints appropriate for Office Open\nXML spreadsheet content.", "members": { - "abstractmethod": { - "name": "abstractmethod", - "kind": "alias", - "path": "omniread.xlsx.parser_base.abstractmethod", - "signature": "", - "docstring": null - }, - "Generic": { - "name": "Generic", - "kind": "alias", - "path": "omniread.xlsx.parser_base.Generic", - "signature": "", - "docstring": null - }, - "TypeVar": { - "name": "TypeVar", - "kind": "alias", - "path": "omniread.xlsx.parser_base.TypeVar", - "signature": "", - "docstring": null - }, "ContentType": { "name": "ContentType", "kind": "class", "path": "omniread.xlsx.parser_base.ContentType", - "signature": "", + "signature": null, "docstring": "Supported MIME types for extracted content.\n\nNotes:\n **Guarantees:**\n\n - This enum represents the declared or inferred media type of the\n content source.\n - It is primarily used for routing content to the appropriate\n parser or downstream consumer.", "members": { "HTML": { "name": "HTML", "kind": "attribute", "path": "omniread.xlsx.parser_base.ContentType.HTML", - "signature": "", + "signature": null, "docstring": "HTML document content." }, "PDF": { "name": "PDF", "kind": "attribute", "path": "omniread.xlsx.parser_base.ContentType.PDF", - "signature": "", + "signature": null, "docstring": "PDF document content." }, "XLSX": { "name": "XLSX", "kind": "attribute", "path": "omniread.xlsx.parser_base.ContentType.XLSX", - "signature": "", + "signature": null, "docstring": "Office Open XML spreadsheet (xlsx/xlsm) content." }, "CSV": { "name": "CSV", "kind": "attribute", "path": "omniread.xlsx.parser_base.ContentType.CSV", - "signature": "", + "signature": null, "docstring": "Comma-separated-value document content." }, "JSON": { "name": "JSON", "kind": "attribute", "path": "omniread.xlsx.parser_base.ContentType.JSON", - "signature": "", + "signature": null, "docstring": "JSON document content." }, "XML": { "name": "XML", "kind": "attribute", "path": "omniread.xlsx.parser_base.ContentType.XML", - "signature": "", + "signature": null, "docstring": "XML document content." } } @@ -398,35 +321,35 @@ "name": "BaseParser", "kind": "class", "path": "omniread.xlsx.parser_base.BaseParser", - "signature": "", + "signature": "BaseParser(content: Content)", "docstring": "Base interface for all parsers.\n\nNotes:\n **Guarantees:**\n\n - A parser is a self-contained object that owns the `Content` it is\n responsible for interpreting.\n - Consumers may rely on early validation of content compatibility\n and type-stable return values from `parse()`.\n\n **Responsibilities:**\n\n - Implementations must declare supported content types via `supported_types`.\n - Implementations must raise parsing-specific exceptions from `parse()`.\n - Implementations must remain deterministic for a given input.", "members": { "supported_types": { "name": "supported_types", "kind": "attribute", "path": "omniread.xlsx.parser_base.BaseParser.supported_types", - "signature": "", + "signature": null, "docstring": "Set of content types supported by this parser. An empty set indicates that the parser is content-type agnostic." }, "content": { "name": "content", "kind": "attribute", "path": "omniread.xlsx.parser_base.BaseParser.content", - "signature": "", + "signature": null, "docstring": null }, "parse": { "name": "parse", "kind": "function", "path": "omniread.xlsx.parser_base.BaseParser.parse", - "signature": "", + "signature": "parse()", "docstring": "Parse the owned content into structured output.\n\nReturns:\n T:\n Parsed, structured representation.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must fully consume the provided content and\n return a deterministic, structured output." }, "supports": { "name": "supports", "kind": "function", "path": "omniread.xlsx.parser_base.BaseParser.supports", - "signature": "", + "signature": "supports()", "docstring": "Check whether this parser supports the content's type.\n\nReturns:\n bool:\n True if the content type is supported; False otherwise." } } @@ -442,7 +365,7 @@ "name": "XlsxParserBase", "kind": "class", "path": "omniread.xlsx.parser_base.XlsxParserBase", - "signature": "", + "signature": null, "docstring": "Base xlsx parser.\n\nNotes:\n **Responsibilities:**\n\n - This class enforces xlsx content-type compatibility and provides\n the extension point for implementing concrete xlsx parsing\n strategies.\n\n **Constraints:**\n\n - Concrete implementations must define the output type `T` and\n implement the `parse()` method.", "members": { "supported_types": { @@ -456,7 +379,7 @@ "name": "parse", "kind": "function", "path": "omniread.xlsx.parser_base.XlsxParserBase.parse", - "signature": "", + "signature": "parse() -> T", "docstring": "Parse xlsx content into a structured output.\n\nReturns:\n T:\n Parsed representation of type `T`.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation." } } @@ -470,53 +393,39 @@ "signature": null, "docstring": "# Summary\n\nXLSX scraper for OmniRead.\n\nThis module defines the scraper responsible for acquiring raw Office Open\nXML spreadsheet content from a backing store via a configured client.\n\nThe scraper does not interpret or parse the acquired bytes; it wraps them in\nthe canonical `Content` model.", "members": { - "Mapping": { - "name": "Mapping", - "kind": "alias", - "path": "omniread.xlsx.scraper.Mapping", - "signature": "", - "docstring": null - }, - "Any": { - "name": "Any", - "kind": "alias", - "path": "omniread.xlsx.scraper.Any", - "signature": "", - "docstring": null - }, "Content": { "name": "Content", "kind": "class", "path": "omniread.xlsx.scraper.Content", - "signature": "", + "signature": "Content(raw: bytes, source: str, content_type: ContentType | None = ..., metadata: Mapping[str, Any] | None = ...)", "docstring": "Normalized representation of extracted content.\n\nNotes:\n **Responsibilities:**\n\n - A `Content` instance represents a raw content payload along with\n minimal contextual metadata describing its origin and type.\n - This class is the primary exchange format between scrapers,\n parsers, and downstream consumers.", "members": { "raw": { "name": "raw", "kind": "attribute", "path": "omniread.xlsx.scraper.Content.raw", - "signature": "", + "signature": null, "docstring": "Raw content bytes as retrieved from the source." }, "source": { "name": "source", "kind": "attribute", "path": "omniread.xlsx.scraper.Content.source", - "signature": "", + "signature": null, "docstring": "Identifier of the content origin (URL, file path, or logical name)." }, "content_type": { "name": "content_type", "kind": "attribute", "path": "omniread.xlsx.scraper.Content.content_type", - "signature": "", + "signature": null, "docstring": "Optional MIME type of the content, if known." }, "metadata": { "name": "metadata", "kind": "attribute", "path": "omniread.xlsx.scraper.Content.metadata", - "signature": "", + "signature": null, "docstring": "Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes)." } } @@ -525,49 +434,49 @@ "name": "ContentType", "kind": "class", "path": "omniread.xlsx.scraper.ContentType", - "signature": "", + "signature": null, "docstring": "Supported MIME types for extracted content.\n\nNotes:\n **Guarantees:**\n\n - This enum represents the declared or inferred media type of the\n content source.\n - It is primarily used for routing content to the appropriate\n parser or downstream consumer.", "members": { "HTML": { "name": "HTML", "kind": "attribute", "path": "omniread.xlsx.scraper.ContentType.HTML", - "signature": "", + "signature": null, "docstring": "HTML document content." }, "PDF": { "name": "PDF", "kind": "attribute", "path": "omniread.xlsx.scraper.ContentType.PDF", - "signature": "", + "signature": null, "docstring": "PDF document content." }, "XLSX": { "name": "XLSX", "kind": "attribute", "path": "omniread.xlsx.scraper.ContentType.XLSX", - "signature": "", + "signature": null, "docstring": "Office Open XML spreadsheet (xlsx/xlsm) content." }, "CSV": { "name": "CSV", "kind": "attribute", "path": "omniread.xlsx.scraper.ContentType.CSV", - "signature": "", + "signature": null, "docstring": "Comma-separated-value document content." }, "JSON": { "name": "JSON", "kind": "attribute", "path": "omniread.xlsx.scraper.ContentType.JSON", - "signature": "", + "signature": null, "docstring": "JSON document content." }, "XML": { "name": "XML", "kind": "attribute", "path": "omniread.xlsx.scraper.ContentType.XML", - "signature": "", + "signature": null, "docstring": "XML document content." } } @@ -576,14 +485,14 @@ "name": "BaseXlsxClient", "kind": "class", "path": "omniread.xlsx.scraper.BaseXlsxClient", - "signature": "", + "signature": null, "docstring": "Abstract client responsible for retrieving spreadsheet bytes.\n\nRetrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).\n\nNotes:\n **Responsibilities:**\n\n - Implementations must accept a source identifier appropriate to\n the backing store.\n - Return the full xlsx binary payload.\n - Raise retrieval-specific errors on failure.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.xlsx.scraper.BaseXlsxClient.fetch", - "signature": "", + "signature": "fetch(source: Any)", "docstring": "Fetch raw xlsx bytes from the given source.\n\nArgs:\n source (Any):\n Identifier of the spreadsheet location, such as a file path,\n object storage key, or remote reference.\n\nReturns:\n bytes:\n Raw xlsx bytes.\n\nRaises:\n Exception:\n Retrieval-specific errors defined by the implementation." } } @@ -592,14 +501,14 @@ "name": "XlsxScraper", "kind": "class", "path": "omniread.xlsx.scraper.XlsxScraper", - "signature": "", + "signature": "XlsxScraper(*, client: BaseXlsxClient)", "docstring": "Scraper for xlsx spreadsheet documents.\n\nNotes:\n **Responsibilities:**\n\n - Fetch raw xlsx bytes via the configured client.\n - Wrap the payload in a canonical `Content` instance with the\n XLSX content type and source identifier.\n\n **Constraints:**\n\n - The scraper does not perform parsing or interpretation.\n - Does not assume a specific storage backend.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.xlsx.scraper.XlsxScraper.fetch", - "signature": "", + "signature": "fetch(source: Any, *, metadata: Mapping[str, Any] | None = None) -> Content", "docstring": "Fetch an xlsx document from the given source.\n\nArgs:\n source (Any):\n Identifier of the spreadsheet source as understood by the\n configured client.\n metadata (Mapping[str, Any] | None, optional):\n Optional metadata to attach to the returned content.\n\nReturns:\n Content:\n A `Content` instance containing raw xlsx bytes, source\n identifier, XLSX content type, and optional metadata.\n\nRaises:\n Exception:\n Retrieval-specific errors raised by the client." } } diff --git a/mcp/omniread/modules/omniread.xlsx.parser.json b/mcp/omniread/modules/omniread.xlsx.parser.json index 38c958a..3621e93 100644 --- a/mcp/omniread/modules/omniread.xlsx.parser.json +++ b/mcp/omniread/modules/omniread.xlsx.parser.json @@ -4,67 +4,39 @@ "path": "omniread.xlsx.parser", "docstring": "# Summary\n\nXLSX parser implementations for OmniRead.\n\nThis module provides a concrete, generic parser for Office Open XML\nspreadsheets. It exposes workbook sheets as lists of string rows so\ndownstream consumers can interpret tabular content without depending on\nopenpyxl directly.\n\nThe parser is intentionally statement-agnostic: it performs no header\ndetection or column interpretation beyond basic cell normalization.", "objects": { - "datetime": { - "name": "datetime", - "kind": "alias", - "path": "omniread.xlsx.parser.datetime", - "signature": "", - "docstring": null - }, - "BytesIO": { - "name": "BytesIO", - "kind": "alias", - "path": "omniread.xlsx.parser.BytesIO", - "signature": "", - "docstring": null - }, - "Any": { - "name": "Any", - "kind": "alias", - "path": "omniread.xlsx.parser.Any", - "signature": "", - "docstring": null - }, - "openpyxl": { - "name": "openpyxl", - "kind": "alias", - "path": "omniread.xlsx.parser.openpyxl", - "signature": "", - "docstring": null - }, "Content": { "name": "Content", "kind": "class", "path": "omniread.xlsx.parser.Content", - "signature": "", + "signature": "Content(raw: bytes, source: str, content_type: ContentType | None = ..., metadata: Mapping[str, Any] | None = ...)", "docstring": "Normalized representation of extracted content.\n\nNotes:\n **Responsibilities:**\n\n - A `Content` instance represents a raw content payload along with\n minimal contextual metadata describing its origin and type.\n - This class is the primary exchange format between scrapers,\n parsers, and downstream consumers.", "members": { "raw": { "name": "raw", "kind": "attribute", "path": "omniread.xlsx.parser.Content.raw", - "signature": "", + "signature": null, "docstring": "Raw content bytes as retrieved from the source." }, "source": { "name": "source", "kind": "attribute", "path": "omniread.xlsx.parser.Content.source", - "signature": "", + "signature": null, "docstring": "Identifier of the content origin (URL, file path, or logical name)." }, "content_type": { "name": "content_type", "kind": "attribute", "path": "omniread.xlsx.parser.Content.content_type", - "signature": "", + "signature": null, "docstring": "Optional MIME type of the content, if known." }, "metadata": { "name": "metadata", "kind": "attribute", "path": "omniread.xlsx.parser.Content.metadata", - "signature": "", + "signature": null, "docstring": "Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes)." } } @@ -73,21 +45,21 @@ "name": "XlsxParserBase", "kind": "class", "path": "omniread.xlsx.parser.XlsxParserBase", - "signature": "", + "signature": null, "docstring": "Base xlsx parser.\n\nNotes:\n **Responsibilities:**\n\n - This class enforces xlsx content-type compatibility and provides\n the extension point for implementing concrete xlsx parsing\n strategies.\n\n **Constraints:**\n\n - Concrete implementations must define the output type `T` and\n implement the `parse()` method.", "members": { "supported_types": { "name": "supported_types", "kind": "attribute", "path": "omniread.xlsx.parser.XlsxParserBase.supported_types", - "signature": "", + "signature": null, "docstring": "Set of content types supported by this parser (XLSX only)." }, "parse": { "name": "parse", "kind": "function", "path": "omniread.xlsx.parser.XlsxParserBase.parse", - "signature": "", + "signature": "parse()", "docstring": "Parse xlsx content into a structured output.\n\nReturns:\n T:\n Parsed representation of type `T`.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation." } } @@ -96,7 +68,7 @@ "name": "XlsxParser", "kind": "class", "path": "omniread.xlsx.parser.XlsxParser", - "signature": "", + "signature": "XlsxParser(content: Content, *, data_only: bool = True, read_only: bool = True)", "docstring": "Generic xlsx parser producing string rows from a worksheet.\n\nNotes:\n **Responsibilities:**\n\n - Lazily load the workbook owned by the parser's content.\n - Normalize cells (including dates and numeric values) into\n deterministic string representations.\n - Expose sheet discovery and row extraction helpers.\n\n **Constraints:**\n\n - Cells are rendered with ``str(value)`` after trimming; date and\n datetime values are rendered in ISO format. Consumers requiring\n locale-specific formatting must convert on their side.", "members": { "workbook": { @@ -117,14 +89,14 @@ "name": "parse", "kind": "function", "path": "omniread.xlsx.parser.XlsxParser.parse", - "signature": "", + "signature": "parse() -> list[list[str]]", "docstring": "Parse the first worksheet into normalized string rows.\n\nReturns:\n list[list[str]]:\n Rows of the default (first) worksheet." }, "rows": { "name": "rows", "kind": "function", "path": "omniread.xlsx.parser.XlsxParser.rows", - "signature": "", + "signature": "rows(sheet: int | str | None = None, *, skip_empty: bool = True) -> list[list[str]]", "docstring": "Extract normalized string rows from a worksheet.\n\nArgs:\n sheet (int | str | None):\n Worksheet index or title; defaults to the first worksheet.\n skip_empty (bool):\n When True (default), rows whose cells are all blank are omitted.\n\nReturns:\n list[list[str]]:\n Normalized rows; trailing blank cells are trimmed per row.\n\nRaises:\n ValueError:\n If the requested sheet does not exist." } } diff --git a/mcp/omniread/modules/omniread.xlsx.parser_base.json b/mcp/omniread/modules/omniread.xlsx.parser_base.json index 598340d..200e04a 100644 --- a/mcp/omniread/modules/omniread.xlsx.parser_base.json +++ b/mcp/omniread/modules/omniread.xlsx.parser_base.json @@ -4,74 +4,53 @@ "path": "omniread.xlsx.parser_base", "docstring": "# Summary\n\nXLSX parser base implementation for OmniRead.\n\nThis module defines the **XLSX-specific parser contract**, extending the\nformat-agnostic `BaseParser` with constraints appropriate for Office Open\nXML spreadsheet content.", "objects": { - "abstractmethod": { - "name": "abstractmethod", - "kind": "alias", - "path": "omniread.xlsx.parser_base.abstractmethod", - "signature": "", - "docstring": null - }, - "Generic": { - "name": "Generic", - "kind": "alias", - "path": "omniread.xlsx.parser_base.Generic", - "signature": "", - "docstring": null - }, - "TypeVar": { - "name": "TypeVar", - "kind": "alias", - "path": "omniread.xlsx.parser_base.TypeVar", - "signature": "", - "docstring": null - }, "ContentType": { "name": "ContentType", "kind": "class", "path": "omniread.xlsx.parser_base.ContentType", - "signature": "", + "signature": null, "docstring": "Supported MIME types for extracted content.\n\nNotes:\n **Guarantees:**\n\n - This enum represents the declared or inferred media type of the\n content source.\n - It is primarily used for routing content to the appropriate\n parser or downstream consumer.", "members": { "HTML": { "name": "HTML", "kind": "attribute", "path": "omniread.xlsx.parser_base.ContentType.HTML", - "signature": "", + "signature": null, "docstring": "HTML document content." }, "PDF": { "name": "PDF", "kind": "attribute", "path": "omniread.xlsx.parser_base.ContentType.PDF", - "signature": "", + "signature": null, "docstring": "PDF document content." }, "XLSX": { "name": "XLSX", "kind": "attribute", "path": "omniread.xlsx.parser_base.ContentType.XLSX", - "signature": "", + "signature": null, "docstring": "Office Open XML spreadsheet (xlsx/xlsm) content." }, "CSV": { "name": "CSV", "kind": "attribute", "path": "omniread.xlsx.parser_base.ContentType.CSV", - "signature": "", + "signature": null, "docstring": "Comma-separated-value document content." }, "JSON": { "name": "JSON", "kind": "attribute", "path": "omniread.xlsx.parser_base.ContentType.JSON", - "signature": "", + "signature": null, "docstring": "JSON document content." }, "XML": { "name": "XML", "kind": "attribute", "path": "omniread.xlsx.parser_base.ContentType.XML", - "signature": "", + "signature": null, "docstring": "XML document content." } } @@ -80,35 +59,35 @@ "name": "BaseParser", "kind": "class", "path": "omniread.xlsx.parser_base.BaseParser", - "signature": "", + "signature": "BaseParser(content: Content)", "docstring": "Base interface for all parsers.\n\nNotes:\n **Guarantees:**\n\n - A parser is a self-contained object that owns the `Content` it is\n responsible for interpreting.\n - Consumers may rely on early validation of content compatibility\n and type-stable return values from `parse()`.\n\n **Responsibilities:**\n\n - Implementations must declare supported content types via `supported_types`.\n - Implementations must raise parsing-specific exceptions from `parse()`.\n - Implementations must remain deterministic for a given input.", "members": { "supported_types": { "name": "supported_types", "kind": "attribute", "path": "omniread.xlsx.parser_base.BaseParser.supported_types", - "signature": "", + "signature": null, "docstring": "Set of content types supported by this parser. An empty set indicates that the parser is content-type agnostic." }, "content": { "name": "content", "kind": "attribute", "path": "omniread.xlsx.parser_base.BaseParser.content", - "signature": "", + "signature": null, "docstring": null }, "parse": { "name": "parse", "kind": "function", "path": "omniread.xlsx.parser_base.BaseParser.parse", - "signature": "", + "signature": "parse()", "docstring": "Parse the owned content into structured output.\n\nReturns:\n T:\n Parsed, structured representation.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation.\n\nNotes:\n **Responsibilities:**\n\n - Implementations must fully consume the provided content and\n return a deterministic, structured output." }, "supports": { "name": "supports", "kind": "function", "path": "omniread.xlsx.parser_base.BaseParser.supports", - "signature": "", + "signature": "supports()", "docstring": "Check whether this parser supports the content's type.\n\nReturns:\n bool:\n True if the content type is supported; False otherwise." } } @@ -124,7 +103,7 @@ "name": "XlsxParserBase", "kind": "class", "path": "omniread.xlsx.parser_base.XlsxParserBase", - "signature": "", + "signature": null, "docstring": "Base xlsx parser.\n\nNotes:\n **Responsibilities:**\n\n - This class enforces xlsx content-type compatibility and provides\n the extension point for implementing concrete xlsx parsing\n strategies.\n\n **Constraints:**\n\n - Concrete implementations must define the output type `T` and\n implement the `parse()` method.", "members": { "supported_types": { @@ -138,7 +117,7 @@ "name": "parse", "kind": "function", "path": "omniread.xlsx.parser_base.XlsxParserBase.parse", - "signature": "", + "signature": "parse() -> T", "docstring": "Parse xlsx content into a structured output.\n\nReturns:\n T:\n Parsed representation of type `T`.\n\nRaises:\n Exception:\n Parsing-specific errors as defined by the implementation." } } diff --git a/mcp/omniread/modules/omniread.xlsx.scraper.json b/mcp/omniread/modules/omniread.xlsx.scraper.json index e87cd15..7e5cdac 100644 --- a/mcp/omniread/modules/omniread.xlsx.scraper.json +++ b/mcp/omniread/modules/omniread.xlsx.scraper.json @@ -4,53 +4,39 @@ "path": "omniread.xlsx.scraper", "docstring": "# Summary\n\nXLSX scraper for OmniRead.\n\nThis module defines the scraper responsible for acquiring raw Office Open\nXML spreadsheet content from a backing store via a configured client.\n\nThe scraper does not interpret or parse the acquired bytes; it wraps them in\nthe canonical `Content` model.", "objects": { - "Mapping": { - "name": "Mapping", - "kind": "alias", - "path": "omniread.xlsx.scraper.Mapping", - "signature": "", - "docstring": null - }, - "Any": { - "name": "Any", - "kind": "alias", - "path": "omniread.xlsx.scraper.Any", - "signature": "", - "docstring": null - }, "Content": { "name": "Content", "kind": "class", "path": "omniread.xlsx.scraper.Content", - "signature": "", + "signature": "Content(raw: bytes, source: str, content_type: ContentType | None = ..., metadata: Mapping[str, Any] | None = ...)", "docstring": "Normalized representation of extracted content.\n\nNotes:\n **Responsibilities:**\n\n - A `Content` instance represents a raw content payload along with\n minimal contextual metadata describing its origin and type.\n - This class is the primary exchange format between scrapers,\n parsers, and downstream consumers.", "members": { "raw": { "name": "raw", "kind": "attribute", "path": "omniread.xlsx.scraper.Content.raw", - "signature": "", + "signature": null, "docstring": "Raw content bytes as retrieved from the source." }, "source": { "name": "source", "kind": "attribute", "path": "omniread.xlsx.scraper.Content.source", - "signature": "", + "signature": null, "docstring": "Identifier of the content origin (URL, file path, or logical name)." }, "content_type": { "name": "content_type", "kind": "attribute", "path": "omniread.xlsx.scraper.Content.content_type", - "signature": "", + "signature": null, "docstring": "Optional MIME type of the content, if known." }, "metadata": { "name": "metadata", "kind": "attribute", "path": "omniread.xlsx.scraper.Content.metadata", - "signature": "", + "signature": null, "docstring": "Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes)." } } @@ -59,49 +45,49 @@ "name": "ContentType", "kind": "class", "path": "omniread.xlsx.scraper.ContentType", - "signature": "", + "signature": null, "docstring": "Supported MIME types for extracted content.\n\nNotes:\n **Guarantees:**\n\n - This enum represents the declared or inferred media type of the\n content source.\n - It is primarily used for routing content to the appropriate\n parser or downstream consumer.", "members": { "HTML": { "name": "HTML", "kind": "attribute", "path": "omniread.xlsx.scraper.ContentType.HTML", - "signature": "", + "signature": null, "docstring": "HTML document content." }, "PDF": { "name": "PDF", "kind": "attribute", "path": "omniread.xlsx.scraper.ContentType.PDF", - "signature": "", + "signature": null, "docstring": "PDF document content." }, "XLSX": { "name": "XLSX", "kind": "attribute", "path": "omniread.xlsx.scraper.ContentType.XLSX", - "signature": "", + "signature": null, "docstring": "Office Open XML spreadsheet (xlsx/xlsm) content." }, "CSV": { "name": "CSV", "kind": "attribute", "path": "omniread.xlsx.scraper.ContentType.CSV", - "signature": "", + "signature": null, "docstring": "Comma-separated-value document content." }, "JSON": { "name": "JSON", "kind": "attribute", "path": "omniread.xlsx.scraper.ContentType.JSON", - "signature": "", + "signature": null, "docstring": "JSON document content." }, "XML": { "name": "XML", "kind": "attribute", "path": "omniread.xlsx.scraper.ContentType.XML", - "signature": "", + "signature": null, "docstring": "XML document content." } } @@ -110,14 +96,14 @@ "name": "BaseXlsxClient", "kind": "class", "path": "omniread.xlsx.scraper.BaseXlsxClient", - "signature": "", + "signature": null, "docstring": "Abstract client responsible for retrieving spreadsheet bytes.\n\nRetrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).\n\nNotes:\n **Responsibilities:**\n\n - Implementations must accept a source identifier appropriate to\n the backing store.\n - Return the full xlsx binary payload.\n - Raise retrieval-specific errors on failure.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.xlsx.scraper.BaseXlsxClient.fetch", - "signature": "", + "signature": "fetch(source: Any)", "docstring": "Fetch raw xlsx bytes from the given source.\n\nArgs:\n source (Any):\n Identifier of the spreadsheet location, such as a file path,\n object storage key, or remote reference.\n\nReturns:\n bytes:\n Raw xlsx bytes.\n\nRaises:\n Exception:\n Retrieval-specific errors defined by the implementation." } } @@ -126,14 +112,14 @@ "name": "XlsxScraper", "kind": "class", "path": "omniread.xlsx.scraper.XlsxScraper", - "signature": "", + "signature": "XlsxScraper(*, client: BaseXlsxClient)", "docstring": "Scraper for xlsx spreadsheet documents.\n\nNotes:\n **Responsibilities:**\n\n - Fetch raw xlsx bytes via the configured client.\n - Wrap the payload in a canonical `Content` instance with the\n XLSX content type and source identifier.\n\n **Constraints:**\n\n - The scraper does not perform parsing or interpretation.\n - Does not assume a specific storage backend.", "members": { "fetch": { "name": "fetch", "kind": "function", "path": "omniread.xlsx.scraper.XlsxScraper.fetch", - "signature": "", + "signature": "fetch(source: Any, *, metadata: Mapping[str, Any] | None = None) -> Content", "docstring": "Fetch an xlsx document from the given source.\n\nArgs:\n source (Any):\n Identifier of the spreadsheet source as understood by the\n configured client.\n metadata (Mapping[str, Any] | None, optional):\n Optional metadata to attach to the returned content.\n\nReturns:\n Content:\n A `Content` instance containing raw xlsx bytes, source\n identifier, XLSX content type, and optional metadata.\n\nRaises:\n Exception:\n Retrieval-specific errors raised by the client." } } diff --git a/omniread/lib/404.html b/omniread/lib/404.html index 501b852..9d7776c 100644 --- a/omniread/lib/404.html +++ b/omniread/lib/404.html @@ -634,6 +634,340 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + + + +
  • + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + + + +
  • + + + diff --git a/omniread/lib/core/content/index.html b/omniread/lib/core/content/index.html index 08a22b3..093f9cb 100644 --- a/omniread/lib/core/content/index.html +++ b/omniread/lib/core/content/index.html @@ -874,6 +874,340 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + + + +
  • + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + + + +
  • + + + diff --git a/omniread/lib/core/index.html b/omniread/lib/core/index.html index aa0cd67..823790d 100644 --- a/omniread/lib/core/index.html +++ b/omniread/lib/core/index.html @@ -643,6 +643,340 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + + + +
  • + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + + + +
  • + + + @@ -1098,7 +1432,7 @@ are safe for downstream consumers to depend on.

    content - Content + Content
    @@ -1437,7 +1771,7 @@ are safe for downstream consumers to depend on.

    Content - Content + Content
    diff --git a/omniread/lib/core/parser/index.html b/omniread/lib/core/parser/index.html index fc5f195..7670201 100644 --- a/omniread/lib/core/parser/index.html +++ b/omniread/lib/core/parser/index.html @@ -796,6 +796,340 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + + + +
  • + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + + + +
  • + + +
    @@ -1048,7 +1382,7 @@ raw content into structured, typed representations.

    content - Content + Content
    diff --git a/omniread/lib/core/scraper/index.html b/omniread/lib/core/scraper/index.html index 5616c4a..d266c5d 100644 --- a/omniread/lib/core/scraper/index.html +++ b/omniread/lib/core/scraper/index.html @@ -763,6 +763,340 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + + + +
  • + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + + + +
  • + + +
    @@ -1064,7 +1398,7 @@ acquiring raw content from external sources.

    Content - Content + Content
    diff --git a/omniread/lib/csv/client/index.html b/omniread/lib/csv/client/index.html index a12b309..1ae9fdb 100644 --- a/omniread/lib/csv/client/index.html +++ b/omniread/lib/csv/client/index.html @@ -9,6 +9,10 @@ + + + + @@ -639,6 +643,499 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + +
  • + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + + + +
  • + + +
    @@ -1003,7 +1500,7 @@ object storage key, or remote reference.

    - Bases: BaseCsvClient

    + Bases: BaseCsvClient

    CSV client that reads from the local filesystem.

    diff --git a/omniread/lib/csv/index.html b/omniread/lib/csv/index.html index d2e3291..046d261 100644 --- a/omniread/lib/csv/index.html +++ b/omniread/lib/csv/index.html @@ -9,6 +9,10 @@ + + + + @@ -639,6 +643,340 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + +
  • + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + + + +
  • + + +
    @@ -1193,7 +1531,7 @@ object storage key, or remote reference.

    - Bases: CsvParserBase[list[list[str]]]

    + Bases: CsvParserBase[list[list[str]]]

    Generic csv parser producing string rows from the document.

    @@ -1239,7 +1577,7 @@ object storage key, or remote reference.

    content - Content + Content
    @@ -1470,7 +1808,7 @@ Normalized rows; trailing blank cells are trimmed per row.

    - Bases: BaseParser[T], Generic[T]

    + Bases: BaseParser[T], Generic[T]

    Base csv parser.

    @@ -1508,7 +1846,7 @@ Normalized rows; trailing blank cells are trimmed per row.

    content - Content + Content
    @@ -1751,7 +2089,7 @@ Normalized rows; trailing blank cells are trimmed per row.

    client - BaseCsvClient + BaseCsvClient
    @@ -1857,7 +2195,7 @@ configured client.

    Content - Content + Content
    @@ -1917,7 +2255,7 @@ identifier, CSV content type, and optional metadata.

    - Bases: BaseCsvClient

    + Bases: BaseCsvClient

    CSV client that reads from the local filesystem.

    diff --git a/omniread/lib/csv/parser/index.html b/omniread/lib/csv/parser/index.html index efbd775..b5ed8ef 100644 --- a/omniread/lib/csv/parser/index.html +++ b/omniread/lib/csv/parser/index.html @@ -9,6 +9,10 @@ + + + + @@ -639,6 +643,502 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + +
  • + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + + + +
  • + + +
    @@ -851,7 +1351,7 @@ delimiter detection.

    - Bases: CsvParserBase[list[list[str]]]

    + Bases: CsvParserBase[list[list[str]]]

    Generic csv parser producing string rows from the document.

    @@ -897,7 +1397,7 @@ delimiter detection.

    content - Content + Content
    diff --git a/omniread/lib/csv/parser_base/index.html b/omniread/lib/csv/parser_base/index.html index 77751c3..b27d867 100644 --- a/omniread/lib/csv/parser_base/index.html +++ b/omniread/lib/csv/parser_base/index.html @@ -9,6 +9,10 @@ + + + + @@ -639,6 +643,493 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + +
  • + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + + + +
  • + + +
    @@ -838,7 +1329,7 @@ comma-separated-value documents.

    - Bases: BaseParser[T], Generic[T]

    + Bases: BaseParser[T], Generic[T]

    Base csv parser.

    @@ -876,7 +1367,7 @@ comma-separated-value documents.

    content - Content + Content
    diff --git a/omniread/lib/csv/scraper/index.html b/omniread/lib/csv/scraper/index.html index 08b2121..114b059 100644 --- a/omniread/lib/csv/scraper/index.html +++ b/omniread/lib/csv/scraper/index.html @@ -9,6 +9,10 @@ + + + + @@ -639,6 +643,460 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + +
  • + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + + + +
  • + + +
    @@ -843,7 +1301,7 @@ the canonical Content model.

    client - BaseCsvClient + BaseCsvClient
    @@ -949,7 +1407,7 @@ configured client.

    Content - Content + Content
    diff --git a/omniread/lib/html/index.html b/omniread/lib/html/index.html index 7aeea9a..cffb505 100644 --- a/omniread/lib/html/index.html +++ b/omniread/lib/html/index.html @@ -643,6 +643,340 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + + + +
  • + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + + + +
  • + + +
    @@ -954,7 +1288,7 @@ use this package only when HTML-specific behavior is required.

    - Bases: BaseParser[T], Generic[T]

    + Bases: BaseParser[T], Generic[T]

    Base HTML parser.

    @@ -1001,7 +1335,7 @@ use this package only when HTML-specific behavior is required.

    content - Content + Content
    @@ -1515,7 +1849,7 @@ A list of rows, where each row is a list of cell text values.

    - Bases: BaseScraper

    + Bases: BaseScraper

    Base HTML scraper using httpx.

    @@ -1704,7 +2038,7 @@ A list of rows, where each row is a list of cell text values.

    Content - Content + Content
    diff --git a/omniread/lib/html/parser/index.html b/omniread/lib/html/parser/index.html index 34e6a20..674c7ae 100644 --- a/omniread/lib/html/parser/index.html +++ b/omniread/lib/html/parser/index.html @@ -832,6 +832,340 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + + + +
  • + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + + + +
  • + + +
    @@ -1074,7 +1408,7 @@ to return a structured representation appropriate for their use case.

    - Bases: BaseParser[T], Generic[T]

    + Bases: BaseParser[T], Generic[T]

    Base HTML parser.

    @@ -1121,7 +1455,7 @@ to return a structured representation appropriate for their use case.

    content - Content + Content
    diff --git a/omniread/lib/html/scraper/index.html b/omniread/lib/html/scraper/index.html index cb7691b..3659a79 100644 --- a/omniread/lib/html/scraper/index.html +++ b/omniread/lib/html/scraper/index.html @@ -772,6 +772,340 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + + + +
  • + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + + + +
  • + + +
    @@ -971,7 +1305,7 @@ layer.

    - Bases: BaseScraper

    + Bases: BaseScraper

    Base HTML scraper using httpx.

    @@ -1160,7 +1494,7 @@ layer.

    Content - Content + Content
    diff --git a/omniread/lib/index.html b/omniread/lib/index.html index cac57a1..176ccc2 100644 --- a/omniread/lib/index.html +++ b/omniread/lib/index.html @@ -1220,6 +1220,340 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + + + +
  • + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + + + +
  • + + +
    @@ -2463,7 +2797,7 @@ required.

    - Bases: BaseParser[T], Generic[T]

    + Bases: BaseParser[T], Generic[T]

    Base HTML parser.

    @@ -2510,7 +2844,7 @@ required.

    content - Content + Content
    @@ -3024,7 +3358,7 @@ A list of rows, where each row is a list of cell text values.

    - Bases: BaseScraper

    + Bases: BaseScraper

    Base HTML scraper using httpx.

    @@ -3213,7 +3547,7 @@ A list of rows, where each row is a list of cell text values.

    Content - Content + Content
    @@ -3354,7 +3688,7 @@ A list of rows, where each row is a list of cell text values.

    - Bases: BaseParser[T], Generic[T]

    + Bases: BaseParser[T], Generic[T]

    Base PDF parser.

    @@ -3390,7 +3724,7 @@ A list of rows, where each row is a list of cell text values.

    content - Content + Content
    @@ -3606,7 +3940,7 @@ A list of rows, where each row is a list of cell text values.

    - Bases: BaseScraper

    + Bases: BaseScraper

    Scraper for PDF sources.

    @@ -3749,7 +4083,7 @@ A list of rows, where each row is a list of cell text values.

    Content - Content + Content
    diff --git a/omniread/lib/objects.inv b/omniread/lib/objects.inv index 65b8cdcbdae40de30ca389195d6510297cfa0932..896710b577ac06e4da33602dabd91ab5fb142528 100644 GIT binary patch delta 1179 zcmV;M1Z4Y*3x*1idw-=`L2lbH5WM#l2AXRfw8tKsA}ETq4d9^2%}`_oF(O-nC_BmT zE6T1UQnTbNb22Q=3|Hjraz(q^?YB*lZ5H+0*Il(al*Qun;^W2Q(}(JOXZbBN#g#c; zmD^(9T3lXc#w@-pa&x@=IKaAZ9@NvgFU{**s@k`XfFJA`u*^@N z>-McKEIMo3Mg;GQHrr%v76mrhOB7?OLz73xygSDgd1EMP=w}qn>D6r!VVi@h--~Q@_VSOvC5mQCIa}6o3O99WwClQZLuRW z1reZLw+OzTKYw)qynHO${K--t5Bt2`R{H>pPfW;}Xj%{YtLTqAVwSc22nOkc*PHE8 zM<}=ZXC1><10AEy9!mN-%%QHUrY$yoZ8Bm)%qb>1#LU3~5Mhxi`k0m5u0|GXzp{$# z&WtfH3=qN!BbJ@P8N#lFkOducbR)L!b3Q^kD52muBFF=8%26nGPlVG{FGfiT z_={sI2Xq@_3Z60*Jums$rfmB8kkKPpg@=OY6@LnnV#s_L#Yv;nWf31vDiv9bHKsyL zR)T~+9JB*29%fpGSzb>0;WI%Yg!)(_n+@2GEbM?QY^fWji5 z7KVm8W{cb%%GuZ2^x64Cs^jGDBATt+d1B5h+<;!5i$(Lh}3Q^@CNhKvHuru}y zxFt*&q2i+8$`03vNQFnE5hz0{nQ+=ANWJ^Xq=Kc;fPX#0hn*9F)9IZ2mn0u5xey)L t#fHJ2Oi+qBG)0m!$vcEiLb#~a>lAr%t2H!iq1vyb%O%Xw`3El;Yii)TF%SR% delta 1313 zcmV++1>X9G3X2Pndw;E)O;6iE5Qgvl6_McDROQ%XQ=zJ&EmBa$Ei#S~5l(D*odEy7 zUMIx%dOqG+bBYqr`|NlV!PW^S+bs$Z{N1XdY@;rue0LxF1-{#*F zmo403k#DkH?FDI`I5)eRrS9a6))dLdi z{>gHl9zR$0TbX%TQrDHP+GcgKPU-|#RLKi&?23Jr;_lp?e8+CtThABEen zF#ki{9&-D=x_wy|Rh_Nddg}C48*3dt=qO1LYfBXNWrrl+ zG$lV0gR*$9ulAa&e$^hK7V<~by&>&o`eL2MHz&J1=?RB zDzr)?s&!ga8QpIAE5VHTpW5z62q#@8M;4$5f()m)7)PYg19Wl}asa8hR5WUvsp+ML zQ`0i7rzYc@P({hMqkxWUOc5#9q9R(ZSw*x=+lolJ1{TuV?l$eBm#H-;3qh=>0M5Wt z!oX8g#DBn&QpSK$QV1u>DP`k`DP}~8DQ9GfDQIL#DQQHBDQacK%~p9{?U~Fy8t_u6 zXD~p34#Gl1`-upqo95)G-u0Tii_xx=3>oII1ZpiqnCn^|CJIg#)a{%uVZD=$1N2e= zFn*%lfteD&&v6-;a=Vr`opQ@0fY?*QHA=oI;(w}maj{w?H8Di23&TIzY`RJF)&i#A ze;;F?12I_@QB0VMn};##7$?&6xO7?{ud_lh$xRgfHZ+`NW+PY=92DD&Mo6!cXW^vU zeP*cTQS|~ZmN-f)oj3_BBxc~pxJMmU9Boz)&86VLpm&8o#0U(v@H&liDImx$_A30# zmw(+X6Ts6uG;vnskVfv@xT?j1O+gxsck!;vO>iS;TFp+%pIZ+XCF}sZ51v#81IV;U zJj3+TOOxF@J?H)(+@&nVcf1UOKdc8Pf3`43Q4Y_uGI>B@T)Q+-bS`Va>*X1{G*v&TIwS2(%Y) zdVq5MIEakq``AKzbqW@d^FXV~cTE&ivM_2IK4n3rLkA>+O0zXg7_r&p4awb-&VT;6 zo^lV>BxZfpC_r>YAnfiC{fxIKOKWG(M=C62;}$2#QEcWoi!;;;;{|NR87wQrGrt-< zCp~HzI-hd7XFv+l5dV(UU?z_5P>pQZcg9f#=@%!JUqiL&OaoKzz9H6RN&qrqqZ3fM zr!1>iyJS1a97M>b9WIXW_ob+wAAgGC*)(3bg@-93AB3_EX9HMfh{$-r+!WKqO52)} z3rkO-rMywk`>cYXau7v{af(=XTLrn`^b}HYF=+V+w?MeUV_^to$b^y8E&?_A&bWeQ zV92jL(}z4g(s3W3e-%13<-(EL(btBdgENApg=!2Eik1mr7vV0|QMNk1=NcOHl=T1F X*!~4?0F_o2z$a*>8F2ms0fICjp+|l> diff --git a/omniread/lib/omniread/core/content/index.html b/omniread/lib/omniread/core/content/index.html deleted file mode 100644 index 3d9956c..0000000 --- a/omniread/lib/omniread/core/content/index.html +++ /dev/null @@ -1,1333 +0,0 @@ - - - - - - - - - - - - - - - - - - - Content - omniread - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -
    - -
    - - - - - - -
    - - -
    - -
    - - - - - - -
    -
    - - - -
    -
    -
    - - - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    - - - - - -

    Content

    - - -
    - - - -

    - omniread.core.content - - -

    - -
    - -

    Summary

    -

    Canonical content models for OmniRead.

    -

    This module defines the format-agnostic content representation used across -all parsers and scrapers in OmniRead.

    -

    The models defined here represent what was extracted, not how it was -retrieved or parsed. Format-specific behavior and metadata must not alter -the semantic meaning of these models.

    - - - -
    - - - - - - -

    Classes

    - -
    - - - -

    - Content - - - - dataclass - - -

    -
    1
    -2
    -3
    -4
    -5
    -6
    Content(
    -    raw: bytes,
    -    source: str,
    -    content_type: ContentType | None = ...,
    -    metadata: Mapping[str, Any] | None = ...,
    -)
    -
    - -
    - - -

    Normalized representation of extracted content.

    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    -3
    -4
    - A `Content` instance represents a raw content payload along with
    -  minimal contextual metadata describing its origin and type.
    -- This class is the primary exchange format between scrapers,
    -  parsers, and downstream consumers.
    -
    -
    - - - -
    - - - - - -
    Attributes
    - -
    - - - -
    - content_type - - - - class-attribute - instance-attribute - - -
    -
    content_type: ContentType | None = None
    -
    - -
    - -

    Optional MIME type of the content, if known.

    -
    - -
    - -
    - - - -
    - metadata - - - - class-attribute - instance-attribute - - -
    -
    metadata: Mapping[str, Any] | None = None
    -
    - -
    - -

    Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes).

    -
    - -
    - -
    - - - -
    - raw - - - - instance-attribute - - -
    -
    raw: bytes
    -
    - -
    - -

    Raw content bytes as retrieved from the source.

    -
    - -
    - -
    - - - -
    - source - - - - instance-attribute - - -
    -
    source: str
    -
    - -
    - -

    Identifier of the content origin (URL, file path, or logical name).

    -
    - -
    - - - - - -
    - -
    - -
    - -
    - - - -

    - ContentType - - -

    - - -
    -

    - Bases: str, Enum

    - - -

    Supported MIME types for extracted content.

    - - -
    - Notes -

    Guarantees:

    -
    1
    -2
    -3
    -4
    - This enum represents the declared or inferred media type of the
    -  content source.
    -- It is primarily used for routing content to the appropriate
    -  parser or downstream consumer.
    -
    -
    - - - -
    - - - - - -
    Attributes
    - -
    - - - -
    - CSV - - - - class-attribute - instance-attribute - - -
    -
    CSV = 'text/csv'
    -
    - -
    - -

    Comma-separated-value document content.

    -
    - -
    - -
    - - - -
    - HTML - - - - class-attribute - instance-attribute - - -
    -
    HTML = 'text/html'
    -
    - -
    - -

    HTML document content.

    -
    - -
    - -
    - - - -
    - JSON - - - - class-attribute - instance-attribute - - -
    -
    JSON = 'application/json'
    -
    - -
    - -

    JSON document content.

    -
    - -
    - -
    - - - -
    - PDF - - - - class-attribute - instance-attribute - - -
    -
    PDF = 'application/pdf'
    -
    - -
    - -

    PDF document content.

    -
    - -
    - -
    - - - -
    - XLSX - - - - class-attribute - instance-attribute - - -
    -
    XLSX = "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet"
    -
    - -
    - -

    Office Open XML spreadsheet (xlsx/xlsm) content.

    -
    - -
    - -
    - - - -
    - XML - - - - class-attribute - instance-attribute - - -
    -
    XML = 'application/xml'
    -
    - -
    - -

    XML document content.

    -
    - -
    - - - - - -
    - -
    - -
    - - - - -
    - -
    - -
    - - - - - - - - - - - - - -
    -
    - - - - - -
    - - - -
    - - - -
    -
    -
    -
    - - - - - - - - - - - - \ No newline at end of file diff --git a/omniread/lib/omniread/core/index.html b/omniread/lib/omniread/core/index.html deleted file mode 100644 index 021e771..0000000 --- a/omniread/lib/omniread/core/index.html +++ /dev/null @@ -1,1921 +0,0 @@ - - - - - - - - - - - - - - - - - - - Core - omniread - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -
    - -
    - - - - - - -
    - - -
    - -
    - - - - - - -
    -
    - - - -
    -
    -
    - - - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    - - - - - -

    Core

    - - -
    - - - -

    - omniread.core - - -

    - -
    - -

    Summary

    -

    Core domain contracts for OmniRead.

    -

    This package defines the format-agnostic domain layer of OmniRead. -It exposes canonical content models and abstract interfaces that are -implemented by format-specific modules (HTML, PDF, etc.).

    -

    Public exports from this package are considered stable contracts and -are safe for downstream consumers to depend on.

    -

    Submodules:

    -
      -
    • content: Canonical content models and enums.
    • -
    • parser: Abstract parsing contracts.
    • -
    • scraper: Abstract scraping contracts.
    • -
    -

    Format-specific behavior must not be introduced at this layer.

    -
    -

    Public API

    -
      -
    • Content
    • -
    • ContentType
    • -
    -
    - - - -
    - - - - - - -

    Classes

    - -
    - - - -

    - BaseParser - - -

    -
    BaseParser(content: Content)
    -
    - -
    -

    - Bases: ABC, Generic[T]

    - - -

    Base interface for all parsers.

    - - -
    - Notes -

    Guarantees:

    -
    1
    -2
    -3
    -4
    - A parser is a self-contained object that owns the `Content` it is
    -  responsible for interpreting.
    -- Consumers may rely on early validation of content compatibility
    -  and type-stable return values from `parse()`.
    -
    -

    Responsibilities:

    -
    1
    -2
    -3
    - Implementations must declare supported content types via `supported_types`.
    -- Implementations must raise parsing-specific exceptions from `parse()`.
    -- Implementations must remain deterministic for a given input.
    -
    -
    -

    Initialize the parser with content to be parsed.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    content - Content - -
    -

    Content instance to be parsed.

    -
    -
    - required -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - ValueError - -
    -

    If the content type is not supported by this parser.

    -
    -
    - - - - -
    - - - - - -
    Attributes
    - -
    - - - -
    - supported_types - - - - class-attribute - instance-attribute - - -
    -
    supported_types: set[ContentType] = set()
    -
    - -
    - -

    Set of content types supported by this parser. An empty set indicates that the parser is content-type agnostic.

    -
    - -
    - -
    Functions
    - -
    - - -
    - parse - - - - abstractmethod - - -
    -
    parse() -> T
    -
    - -
    - -

    Parse the owned content into structured output.

    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    T - T - -
    -

    Parsed, structured representation.

    -
    -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - Exception - -
    -

    Parsing-specific errors as defined by the implementation.

    -
    -
    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    - Implementations must fully consume the provided content and
    -  return a deterministic, structured output.
    -
    -
    -
    - -
    - -
    - - -
    - supports - - -
    -
    supports() -> bool
    -
    - -
    - -

    Check whether this parser supports the content's type.

    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    bool - bool - -
    -

    True if the content type is supported; False otherwise.

    -
    -
    - -
    - -
    - - - -
    - -
    - -
    - -
    - - - -

    - BaseScraper - - -

    - - -
    -

    - Bases: ABC

    - - -

    Base interface for all scrapers.

    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    -3
    -4
    -5
    -6
    -7
    - A scraper is responsible ONLY for fetching raw content (bytes)
    -  from a source. It must not interpret or parse it.
    -- A scraper is a stateless acquisition component that retrieves raw
    -  content from a source and returns it as a `Content` object.
    -- Scrapers define how content is obtained, not what the content means.
    -- Implementations may vary in transport mechanism, authentication
    -  strategy, retry and backoff behavior.
    -
    -

    Constraints:

    -
    1
    -2
    - Implementations must not parse content, modify content semantics,
    -  or couple scraping logic to a specific parser.
    -
    -
    - - - -
    - - - - - - - -
    Functions
    - -
    - - -
    - fetch - - - - abstractmethod - - -
    -
    1
    -2
    -3
    -4
    -5
    fetch(
    -    source: str,
    -    *,
    -    metadata: Mapping[str, Any] | None = None
    -) -> Content
    -
    - -
    - -

    Fetch raw content from the given source.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    source - str - -
    -

    Location identifier (URL, file path, S3 URI, etc.).

    -
    -
    - required -
    metadata - Mapping[str, Any] | None - -
    -

    Optional hints for the scraper (headers, auth, etc.).

    -
    -
    - None -
    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    Content - Content - -
    -

    Content object containing raw bytes and metadata.

    -
    -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - Exception - -
    -

    Retrieval-specific errors as defined by the implementation.

    -
    -
    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    - Implementations must retrieve the content referenced by `source`
    -  and return it as raw bytes wrapped in a `Content` object.
    -
    -
    -
    - -
    - - - -
    - -
    - -
    - -
    - - - -

    - Content - - - - dataclass - - -

    -
    1
    -2
    -3
    -4
    -5
    -6
    Content(
    -    raw: bytes,
    -    source: str,
    -    content_type: ContentType | None = ...,
    -    metadata: Mapping[str, Any] | None = ...,
    -)
    -
    - -
    - - -

    Normalized representation of extracted content.

    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    -3
    -4
    - A `Content` instance represents a raw content payload along with
    -  minimal contextual metadata describing its origin and type.
    -- This class is the primary exchange format between scrapers,
    -  parsers, and downstream consumers.
    -
    -
    - - - -
    - - - - - -
    Attributes
    - -
    - - - -
    - content_type - - - - class-attribute - instance-attribute - - -
    -
    content_type: ContentType | None = None
    -
    - -
    - -

    Optional MIME type of the content, if known.

    -
    - -
    - -
    - - - -
    - metadata - - - - class-attribute - instance-attribute - - -
    -
    metadata: Mapping[str, Any] | None = None
    -
    - -
    - -

    Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes).

    -
    - -
    - -
    - - - -
    - raw - - - - instance-attribute - - -
    -
    raw: bytes
    -
    - -
    - -

    Raw content bytes as retrieved from the source.

    -
    - -
    - -
    - - - -
    - source - - - - instance-attribute - - -
    -
    source: str
    -
    - -
    - -

    Identifier of the content origin (URL, file path, or logical name).

    -
    - -
    - - - - - -
    - -
    - -
    - -
    - - - -

    - ContentType - - -

    - - -
    -

    - Bases: str, Enum

    - - -

    Supported MIME types for extracted content.

    - - -
    - Notes -

    Guarantees:

    -
    1
    -2
    -3
    -4
    - This enum represents the declared or inferred media type of the
    -  content source.
    -- It is primarily used for routing content to the appropriate
    -  parser or downstream consumer.
    -
    -
    - - - -
    - - - - - -
    Attributes
    - -
    - - - -
    - CSV - - - - class-attribute - instance-attribute - - -
    -
    CSV = 'text/csv'
    -
    - -
    - -

    Comma-separated-value document content.

    -
    - -
    - -
    - - - -
    - HTML - - - - class-attribute - instance-attribute - - -
    -
    HTML = 'text/html'
    -
    - -
    - -

    HTML document content.

    -
    - -
    - -
    - - - -
    - JSON - - - - class-attribute - instance-attribute - - -
    -
    JSON = 'application/json'
    -
    - -
    - -

    JSON document content.

    -
    - -
    - -
    - - - -
    - PDF - - - - class-attribute - instance-attribute - - -
    -
    PDF = 'application/pdf'
    -
    - -
    - -

    PDF document content.

    -
    - -
    - -
    - - - -
    - XLSX - - - - class-attribute - instance-attribute - - -
    -
    XLSX = "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet"
    -
    - -
    - -

    Office Open XML spreadsheet (xlsx/xlsm) content.

    -
    - -
    - -
    - - - -
    - XML - - - - class-attribute - instance-attribute - - -
    -
    XML = 'application/xml'
    -
    - -
    - -

    XML document content.

    -
    - -
    - - - - - -
    - -
    - -
    - - - - -
    - -
    - -
    - - - - - - - - - - - - - -
    -
    - - - - - -
    - - - -
    - - - -
    -
    -
    -
    - - - - - - - - - - - - \ No newline at end of file diff --git a/omniread/lib/omniread/core/parser/index.html b/omniread/lib/omniread/core/parser/index.html deleted file mode 100644 index d5f0adf..0000000 --- a/omniread/lib/omniread/core/parser/index.html +++ /dev/null @@ -1,1167 +0,0 @@ - - - - - - - - - - - - - - - - - - - Parser - omniread - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -
    - -
    - - - - - - -
    - - -
    - -
    - - - - - - -
    -
    - - - -
    -
    -
    - - - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    - - - - - -

    Parser

    - - -
    - - - -

    - omniread.core.parser - - -

    - -
    - -

    Summary

    -

    Abstract parsing contracts for OmniRead.

    -

    This module defines the format-agnostic parser interface used to transform -raw content into structured, typed representations.

    -

    Parsers are responsible for:

    -
      -
    • Interpreting a single Content instance
    • -
    • Validating compatibility with the content type
    • -
    • Producing a structured output suitable for downstream consumers
    • -
    -

    Parsers are not responsible for:

    -
      -
    • Fetching or acquiring content
    • -
    • Performing retries or error recovery
    • -
    • Managing multiple content sources
    • -
    - - - -
    - - - - - - -

    Classes

    - -
    - - - -

    - BaseParser - - -

    -
    BaseParser(content: Content)
    -
    - -
    -

    - Bases: ABC, Generic[T]

    - - -

    Base interface for all parsers.

    - - -
    - Notes -

    Guarantees:

    -
    1
    -2
    -3
    -4
    - A parser is a self-contained object that owns the `Content` it is
    -  responsible for interpreting.
    -- Consumers may rely on early validation of content compatibility
    -  and type-stable return values from `parse()`.
    -
    -

    Responsibilities:

    -
    1
    -2
    -3
    - Implementations must declare supported content types via `supported_types`.
    -- Implementations must raise parsing-specific exceptions from `parse()`.
    -- Implementations must remain deterministic for a given input.
    -
    -
    -

    Initialize the parser with content to be parsed.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    content - Content - -
    -

    Content instance to be parsed.

    -
    -
    - required -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - ValueError - -
    -

    If the content type is not supported by this parser.

    -
    -
    - - - - -
    - - - - - -
    Attributes
    - -
    - - - -
    - supported_types - - - - class-attribute - instance-attribute - - -
    -
    supported_types: set[ContentType] = set()
    -
    - -
    - -

    Set of content types supported by this parser. An empty set indicates that the parser is content-type agnostic.

    -
    - -
    - -
    Functions
    - -
    - - -
    - parse - - - - abstractmethod - - -
    -
    parse() -> T
    -
    - -
    - -

    Parse the owned content into structured output.

    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    T - T - -
    -

    Parsed, structured representation.

    -
    -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - Exception - -
    -

    Parsing-specific errors as defined by the implementation.

    -
    -
    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    - Implementations must fully consume the provided content and
    -  return a deterministic, structured output.
    -
    -
    -
    - -
    - -
    - - -
    - supports - - -
    -
    supports() -> bool
    -
    - -
    - -

    Check whether this parser supports the content's type.

    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    bool - bool - -
    -

    True if the content type is supported; False otherwise.

    -
    -
    - -
    - -
    - - - -
    - -
    - -
    - - - - -
    - -
    - -
    - - - - - - - - - - - - - -
    -
    - - - - - -
    - - - -
    - - - -
    -
    -
    -
    - - - - - - - - - - - - \ No newline at end of file diff --git a/omniread/lib/omniread/core/scraper/index.html b/omniread/lib/omniread/core/scraper/index.html deleted file mode 100644 index ba3dd56..0000000 --- a/omniread/lib/omniread/core/scraper/index.html +++ /dev/null @@ -1,1069 +0,0 @@ - - - - - - - - - - - - - - - - - - - Scraper - omniread - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -
    - -
    - - - - - - -
    - - -
    - -
    - - - - - - -
    -
    - - - -
    -
    -
    - - - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    - - - - - -

    Scraper

    - - -
    - - - -

    - omniread.core.scraper - - -

    - -
    - -

    Summary

    -

    Abstract scraping contracts for OmniRead.

    -

    This module defines the format-agnostic scraper interface responsible for -acquiring raw content from external sources.

    -

    Scrapers are responsible for:

    -
      -
    • Locating and retrieving raw content bytes
    • -
    • Attaching minimal contextual metadata
    • -
    • Returning normalized Content objects
    • -
    -

    Scrapers are explicitly NOT responsible for:

    -
      -
    • Parsing or interpreting content
    • -
    • Inferring structure or semantics
    • -
    • Performing content-type specific processing
    • -
    -

    All interpretation must be delegated to parsers.

    - - - -
    - - - - - - -

    Classes

    - -
    - - - -

    - BaseScraper - - -

    - - -
    -

    - Bases: ABC

    - - -

    Base interface for all scrapers.

    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    -3
    -4
    -5
    -6
    -7
    - A scraper is responsible ONLY for fetching raw content (bytes)
    -  from a source. It must not interpret or parse it.
    -- A scraper is a stateless acquisition component that retrieves raw
    -  content from a source and returns it as a `Content` object.
    -- Scrapers define how content is obtained, not what the content means.
    -- Implementations may vary in transport mechanism, authentication
    -  strategy, retry and backoff behavior.
    -
    -

    Constraints:

    -
    1
    -2
    - Implementations must not parse content, modify content semantics,
    -  or couple scraping logic to a specific parser.
    -
    -
    - - - -
    - - - - - - - -
    Functions
    - -
    - - -
    - fetch - - - - abstractmethod - - -
    -
    1
    -2
    -3
    -4
    -5
    fetch(
    -    source: str,
    -    *,
    -    metadata: Mapping[str, Any] | None = None
    -) -> Content
    -
    - -
    - -

    Fetch raw content from the given source.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    source - str - -
    -

    Location identifier (URL, file path, S3 URI, etc.).

    -
    -
    - required -
    metadata - Mapping[str, Any] | None - -
    -

    Optional hints for the scraper (headers, auth, etc.).

    -
    -
    - None -
    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    Content - Content - -
    -

    Content object containing raw bytes and metadata.

    -
    -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - Exception - -
    -

    Retrieval-specific errors as defined by the implementation.

    -
    -
    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    - Implementations must retrieve the content referenced by `source`
    -  and return it as raw bytes wrapped in a `Content` object.
    -
    -
    -
    - -
    - - - -
    - -
    - -
    - - - - -
    - -
    - -
    - - - - - - - - - - - - - -
    -
    - - - - - -
    - - - -
    - - - -
    -
    -
    -
    - - - - - - - - - - - - \ No newline at end of file diff --git a/omniread/lib/omniread/csv/client/index.html b/omniread/lib/omniread/csv/client/index.html deleted file mode 100644 index 5b7311d..0000000 --- a/omniread/lib/omniread/csv/client/index.html +++ /dev/null @@ -1,1216 +0,0 @@ - - - - - - - - - - - - - - - - - - - Client - omniread - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -
    - -
    - - - - - - -
    - - -
    - -
    - - - - - - -
    -
    - - - -
    -
    -
    - - - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    - - - - - -

    Client

    - - -
    - - - -

    - omniread.csv.client - - -

    - -
    - -

    Summary

    -

    CSV client abstractions for OmniRead.

    -

    This module defines the client layer responsible for retrieving raw -comma-separated-value document bytes from a concrete backing store.

    -

    Clients provide low-level access to csv binaries and are intentionally -decoupled from scraping and parsing logic. They do not perform validation, -interpretation, or content extraction.

    -

    Typical backing stores include:

    -
      -
    • Local filesystems
    • -
    • Object storage (S3, GCS, etc.)
    • -
    • Network file systems
    • -
    - - - -
    - - - - - - -

    Classes

    - -
    - - - -

    - BaseCsvClient - - -

    - - -
    -

    - Bases: ABC

    - - -

    Abstract client responsible for retrieving csv bytes.

    -

    Retrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).

    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    -3
    -4
    - Implementations must accept a source identifier appropriate to
    -  the backing store.
    -- Return the full csv binary payload.
    -- Raise retrieval-specific errors on failure.
    -
    -
    - - - -
    - - - - - - - -
    Functions
    - -
    - - -
    - fetch - - - - abstractmethod - - -
    -
    fetch(source: Any) -> bytes
    -
    - -
    - -

    Fetch raw csv bytes from the given source.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    source - Any - -
    -

    Identifier of the csv location, such as a file path, -object storage key, or remote reference.

    -
    -
    - required -
    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    bytes - bytes - -
    -

    Raw csv bytes.

    -
    -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - Exception - -
    -

    Retrieval-specific errors defined by the implementation.

    -
    -
    - -
    - -
    - - - -
    - -
    - -
    - -
    - - - -

    - FileSystemCsvClient - - -

    - - -
    -

    - Bases: BaseCsvClient

    - - -

    CSV client that reads from the local filesystem.

    - - -
    - Notes -

    Guarantees:

    -
    1
    -2
    - This client reads csv files directly from the disk and
    -  returns their raw binary contents.
    -
    -
    - - - -
    - - - - - - - -
    Functions
    - -
    - - -
    - fetch - - -
    -
    fetch(path: Path) -> bytes
    -
    - -
    - -

    Read a csv file from the local filesystem.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    path - Path - -
    -

    Filesystem path to the csv file.

    -
    -
    - required -
    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    bytes - bytes - -
    -

    Raw csv bytes.

    -
    -
    - - -

    Raises:

    - - - - - - - - - - - - - - - - - -
    TypeDescription
    - FileNotFoundError - -
    -

    If the path does not exist.

    -
    -
    - ValueError - -
    -

    If the path exists but is not a file.

    -
    -
    - -
    - -
    - - - -
    - -
    - -
    - - - - -
    - -
    - -
    - - - - - - - - - - - - - -
    -
    - - - - - -
    - - - -
    - - - -
    -
    -
    -
    - - - - - - - - - - - - \ No newline at end of file diff --git a/omniread/lib/omniread/csv/index.html b/omniread/lib/omniread/csv/index.html deleted file mode 100644 index b49c903..0000000 --- a/omniread/lib/omniread/csv/index.html +++ /dev/null @@ -1,2135 +0,0 @@ - - - - - - - - - - - - - - - - - - - Csv - omniread - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -
    - -
    - - - - - - -
    - - -
    - -
    - - - - - - -
    -
    - - - -
    -
    -
    - - - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    - - - - - -

    Csv

    - - -
    - - - -

    - omniread.csv - - -

    - -
    - -

    Summary

    -

    CSV subpackage for OmniRead.

    -

    Provides acquisition and parsing of comma-separated-value content:

    -
      -
    • BaseCsvClient: abstract backing-store client for csv bytes.
    • -
    • FileSystemCsvClient: local filesystem implementation.
    • -
    • CsvScraper: wraps fetched bytes into canonical Content.
    • -
    • CsvParserBase: content-type-enforcing parser contract.
    • -
    • CsvParser: generic string-row parser built on the standard csv module.
    • -
    - - - -
    - - - - - - -

    Classes

    - -
    - - - -

    - BaseCsvClient - - -

    - - -
    -

    - Bases: ABC

    - - -

    Abstract client responsible for retrieving csv bytes.

    -

    Retrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).

    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    -3
    -4
    - Implementations must accept a source identifier appropriate to
    -  the backing store.
    -- Return the full csv binary payload.
    -- Raise retrieval-specific errors on failure.
    -
    -
    - - - -
    - - - - - - - -
    Functions
    - -
    - - -
    - fetch - - - - abstractmethod - - -
    -
    fetch(source: Any) -> bytes
    -
    - -
    - -

    Fetch raw csv bytes from the given source.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    source - Any - -
    -

    Identifier of the csv location, such as a file path, -object storage key, or remote reference.

    -
    -
    - required -
    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    bytes - bytes - -
    -

    Raw csv bytes.

    -
    -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - Exception - -
    -

    Retrieval-specific errors defined by the implementation.

    -
    -
    - -
    - -
    - - - -
    - -
    - -
    - -
    - - - -

    - CsvParser - - -

    -
    CsvParser(content: Content)
    -
    - -
    -

    - Bases: CsvParserBase[list[list[str]]]

    - - -

    Generic csv parser producing string rows from the document.

    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    -3
    -4
    -5
    - Decode the payload (UTF-8 with BOM support, Latin-1 fallback).
    -- Detect the delimiter from a leading sample (`,` `;` tab `|`),
    -  defaulting to `,`.
    -- Normalize cells into deterministic stripped string values.
    -- Expose row extraction helpers mirroring `XlsxParser.rows`.
    -
    -

    Constraints:

    -
    1
    -2
    -3
    -4
    - All values are strings; consumers requiring typed values must
    -  convert on their side.
    -- Quoted fields containing delimiters/newlines are handled by
    -  the standard ``csv`` module.
    -
    -
    -

    Initialize the parser.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    content - Content - -
    -

    CSV content to parse; its type must be supported.

    -
    -
    - required -
    - - - - -
    - - - - - -
    Attributes
    - -
    - - - -
    - supported_types - - - - class-attribute - instance-attribute - - -
    -
    supported_types: set[ContentType] = {CSV}
    -
    - -
    - -

    Set of content types supported by this parser (CSV only).

    -
    - -
    - -
    Functions
    - -
    - - -
    - parse - - -
    -
    parse() -> list[list[str]]
    -
    - -
    - -

    Parse the document into normalized string rows.

    - - -

    Returns:

    - - - - - - - - - - - - - -
    TypeDescription
    - list[list[str]] - -
    -

    list[list[str]]: -Rows of the document.

    -
    -
    - -
    - -
    - -
    - - -
    - rows - - -
    -
    rows(*, skip_empty: bool = True) -> list[list[str]]
    -
    - -
    - -

    Extract normalized string rows from the document.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    skip_empty - bool - -
    -

    When True (default), rows whose cells are all blank are -omitted.

    -
    -
    - True -
    - - -

    Returns:

    - - - - - - - - - - - - - -
    TypeDescription
    - list[list[str]] - -
    -

    list[list[str]]: -Normalized rows; trailing blank cells are trimmed per row.

    -
    -
    - -
    - -
    - -
    - - -
    - supports - - -
    -
    supports() -> bool
    -
    - -
    - -

    Check whether this parser supports the content's type.

    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    bool - bool - -
    -

    True if the content type is supported; False otherwise.

    -
    -
    - -
    - -
    - - - -
    - -
    - -
    - -
    - - - -

    - CsvParserBase - - -

    -
    CsvParserBase(content: Content)
    -
    - -
    -

    - Bases: BaseParser[T], Generic[T]

    - - -

    Base csv parser.

    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    -3
    - This class enforces csv content-type compatibility and provides
    -  the extension point for implementing concrete csv parsing
    -  strategies.
    -
    -

    Constraints:

    -
    1
    -2
    - Concrete implementations must define the output type `T` and
    -  implement the `parse()` method.
    -
    -
    -

    Initialize the parser with content to be parsed.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    content - Content - -
    -

    Content instance to be parsed.

    -
    -
    - required -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - ValueError - -
    -

    If the content type is not supported by this parser.

    -
    -
    - - - - -
    - - - - - -
    Attributes
    - -
    - - - -
    - supported_types - - - - class-attribute - instance-attribute - - -
    -
    supported_types: set[ContentType] = {CSV}
    -
    - -
    - -

    Set of content types supported by this parser (CSV only).

    -
    - -
    - -
    Functions
    - -
    - - -
    - parse - - - - abstractmethod - - -
    -
    parse() -> T
    -
    - -
    - -

    Parse csv content into a structured output.

    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    T - T - -
    -

    Parsed representation of type T.

    -
    -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - Exception - -
    -

    Parsing-specific errors as defined by the implementation.

    -
    -
    - -
    - -
    - -
    - - -
    - supports - - -
    -
    supports() -> bool
    -
    - -
    - -

    Check whether this parser supports the content's type.

    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    bool - bool - -
    -

    True if the content type is supported; False otherwise.

    -
    -
    - -
    - -
    - - - -
    - -
    - -
    - -
    - - - -

    - CsvScraper - - -

    -
    CsvScraper(*, client: BaseCsvClient)
    -
    - -
    - - -

    Scraper for csv documents.

    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    -3
    - Fetch raw csv bytes via the configured client.
    -- Wrap the payload in a canonical `Content` instance with the
    -  CSV content type and source identifier.
    -
    -

    Constraints:

    -
    1
    -2
    - The scraper does not perform parsing or interpretation.
    -- Does not assume a specific storage backend.
    -
    -
    -

    Initialize the CSV scraper.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    client - BaseCsvClient - -
    -

    Client responsible for retrieving raw csv bytes.

    -
    -
    - required -
    - - - - -
    - - - - - - - -
    Functions
    - -
    - - -
    - fetch - - -
    -
    1
    -2
    -3
    -4
    -5
    fetch(
    -    source: Any,
    -    *,
    -    metadata: Mapping[str, Any] | None = None
    -) -> Content
    -
    - -
    - -

    Fetch a csv document from the given source.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    source - Any - -
    -

    Identifier of the csv source as understood by the -configured client.

    -
    -
    - required -
    metadata - Mapping[str, Any] | None - -
    -

    Optional metadata to attach to the returned content.

    -
    -
    - None -
    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    Content - Content - -
    -

    A Content instance containing raw csv bytes, source -identifier, CSV content type, and optional metadata.

    -
    -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - Exception - -
    -

    Retrieval-specific errors raised by the client.

    -
    -
    - -
    - -
    - - - -
    - -
    - -
    - -
    - - - -

    - FileSystemCsvClient - - -

    - - -
    -

    - Bases: BaseCsvClient

    - - -

    CSV client that reads from the local filesystem.

    - - -
    - Notes -

    Guarantees:

    -
    1
    -2
    - This client reads csv files directly from the disk and
    -  returns their raw binary contents.
    -
    -
    - - - -
    - - - - - - - -
    Functions
    - -
    - - -
    - fetch - - -
    -
    fetch(path: Path) -> bytes
    -
    - -
    - -

    Read a csv file from the local filesystem.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    path - Path - -
    -

    Filesystem path to the csv file.

    -
    -
    - required -
    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    bytes - bytes - -
    -

    Raw csv bytes.

    -
    -
    - - -

    Raises:

    - - - - - - - - - - - - - - - - - -
    TypeDescription
    - FileNotFoundError - -
    -

    If the path does not exist.

    -
    -
    - ValueError - -
    -

    If the path exists but is not a file.

    -
    -
    - -
    - -
    - - - -
    - -
    - -
    - - - - -
    - -
    - -
    - - - - - - - - - - - - - -
    -
    - - - - - -
    - - - -
    - - - -
    -
    -
    -
    - - - - - - - - - - - - \ No newline at end of file diff --git a/omniread/lib/omniread/csv/parser/index.html b/omniread/lib/omniread/csv/parser/index.html deleted file mode 100644 index 2bb8400..0000000 --- a/omniread/lib/omniread/csv/parser/index.html +++ /dev/null @@ -1,1188 +0,0 @@ - - - - - - - - - - - - - - - - - - - Parser - omniread - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -
    - -
    - - - - - - -
    - - -
    - -
    - - - - - - -
    -
    - - - -
    -
    -
    - - - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    - - - - - -

    Parser

    - - -
    - - - -

    - omniread.csv.parser - - -

    - -
    - -

    Summary

    -

    CSV parser implementations for OmniRead.

    -

    This module provides a concrete, generic parser for comma-separated-value -documents. It exposes records as lists of string cells so downstream -consumers can interpret tabular content without depending on the csv -module directly.

    -

    The parser is intentionally statement-agnostic: it performs no header -detection or column interpretation beyond basic cell normalization and -delimiter detection.

    - - - -
    - - - - - - -

    Classes

    - -
    - - - -

    - CsvParser - - -

    -
    CsvParser(content: Content)
    -
    - -
    -

    - Bases: CsvParserBase[list[list[str]]]

    - - -

    Generic csv parser producing string rows from the document.

    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    -3
    -4
    -5
    - Decode the payload (UTF-8 with BOM support, Latin-1 fallback).
    -- Detect the delimiter from a leading sample (`,` `;` tab `|`),
    -  defaulting to `,`.
    -- Normalize cells into deterministic stripped string values.
    -- Expose row extraction helpers mirroring `XlsxParser.rows`.
    -
    -

    Constraints:

    -
    1
    -2
    -3
    -4
    - All values are strings; consumers requiring typed values must
    -  convert on their side.
    -- Quoted fields containing delimiters/newlines are handled by
    -  the standard ``csv`` module.
    -
    -
    -

    Initialize the parser.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    content - Content - -
    -

    CSV content to parse; its type must be supported.

    -
    -
    - required -
    - - - - -
    - - - - - -
    Attributes
    - -
    - - - -
    - supported_types - - - - class-attribute - instance-attribute - - -
    -
    supported_types: set[ContentType] = {CSV}
    -
    - -
    - -

    Set of content types supported by this parser (CSV only).

    -
    - -
    - -
    Functions
    - -
    - - -
    - parse - - -
    -
    parse() -> list[list[str]]
    -
    - -
    - -

    Parse the document into normalized string rows.

    - - -

    Returns:

    - - - - - - - - - - - - - -
    TypeDescription
    - list[list[str]] - -
    -

    list[list[str]]: -Rows of the document.

    -
    -
    - -
    - -
    - -
    - - -
    - rows - - -
    -
    rows(*, skip_empty: bool = True) -> list[list[str]]
    -
    - -
    - -

    Extract normalized string rows from the document.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    skip_empty - bool - -
    -

    When True (default), rows whose cells are all blank are -omitted.

    -
    -
    - True -
    - - -

    Returns:

    - - - - - - - - - - - - - -
    TypeDescription
    - list[list[str]] - -
    -

    list[list[str]]: -Normalized rows; trailing blank cells are trimmed per row.

    -
    -
    - -
    - -
    - -
    - - -
    - supports - - -
    -
    supports() -> bool
    -
    - -
    - -

    Check whether this parser supports the content's type.

    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    bool - bool - -
    -

    True if the content type is supported; False otherwise.

    -
    -
    - -
    - -
    - - - -
    - -
    - -
    - - - - -
    - -
    - -
    - - - - - - - - - - - - - -
    -
    - - - - - -
    - - - -
    - - - -
    -
    -
    -
    - - - - - - - - - - - - \ No newline at end of file diff --git a/omniread/lib/omniread/csv/parser_base/index.html b/omniread/lib/omniread/csv/parser_base/index.html deleted file mode 100644 index 87ca2b9..0000000 --- a/omniread/lib/omniread/csv/parser_base/index.html +++ /dev/null @@ -1,1143 +0,0 @@ - - - - - - - - - - - - - - - - - - - Parser Base - omniread - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -
    - -
    - - - - - - -
    - - -
    - -
    - - - - - - -
    -
    - - - -
    -
    -
    - - - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    - - - - - -

    Parser Base

    - - -
    - - - -

    - omniread.csv.parser_base - - -

    - -
    - -

    Summary

    -

    CSV parser base implementation for OmniRead.

    -

    This module defines the CSV-specific parser contract, extending the -format-agnostic BaseParser with constraints appropriate for -comma-separated-value documents.

    - - - -
    - - - - - - -

    Classes

    - -
    - - - -

    - CsvParserBase - - -

    -
    CsvParserBase(content: Content)
    -
    - -
    -

    - Bases: BaseParser[T], Generic[T]

    - - -

    Base csv parser.

    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    -3
    - This class enforces csv content-type compatibility and provides
    -  the extension point for implementing concrete csv parsing
    -  strategies.
    -
    -

    Constraints:

    -
    1
    -2
    - Concrete implementations must define the output type `T` and
    -  implement the `parse()` method.
    -
    -
    -

    Initialize the parser with content to be parsed.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    content - Content - -
    -

    Content instance to be parsed.

    -
    -
    - required -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - ValueError - -
    -

    If the content type is not supported by this parser.

    -
    -
    - - - - -
    - - - - - -
    Attributes
    - -
    - - - -
    - supported_types - - - - class-attribute - instance-attribute - - -
    -
    supported_types: set[ContentType] = {CSV}
    -
    - -
    - -

    Set of content types supported by this parser (CSV only).

    -
    - -
    - -
    Functions
    - -
    - - -
    - parse - - - - abstractmethod - - -
    -
    parse() -> T
    -
    - -
    - -

    Parse csv content into a structured output.

    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    T - T - -
    -

    Parsed representation of type T.

    -
    -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - Exception - -
    -

    Parsing-specific errors as defined by the implementation.

    -
    -
    - -
    - -
    - -
    - - -
    - supports - - -
    -
    supports() -> bool
    -
    - -
    - -

    Check whether this parser supports the content's type.

    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    bool - bool - -
    -

    True if the content type is supported; False otherwise.

    -
    -
    - -
    - -
    - - - -
    - -
    - -
    - - - - -
    - -
    - -
    - - - - - - - - - - - - - -
    -
    - - - - - -
    - - - -
    - - - -
    -
    -
    -
    - - - - - - - - - - - - \ No newline at end of file diff --git a/omniread/lib/omniread/csv/scraper/index.html b/omniread/lib/omniread/csv/scraper/index.html deleted file mode 100644 index 4f02af6..0000000 --- a/omniread/lib/omniread/csv/scraper/index.html +++ /dev/null @@ -1,1070 +0,0 @@ - - - - - - - - - - - - - - - - - - - Scraper - omniread - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -
    - -
    - - - - - - -
    - - -
    - -
    - - - - - - -
    -
    - - - -
    -
    -
    - - - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    - - - - - -

    Scraper

    - - -
    - - - -

    - omniread.csv.scraper - - -

    - -
    - -

    Summary

    -

    CSV scraper for OmniRead.

    -

    This module defines the scraper responsible for acquiring raw -comma-separated-value document content from a backing store via a -configured client.

    -

    The scraper does not interpret or parse the acquired bytes; it wraps them in -the canonical Content model.

    - - - -
    - - - - - - -

    Classes

    - -
    - - - -

    - CsvScraper - - -

    -
    CsvScraper(*, client: BaseCsvClient)
    -
    - -
    - - -

    Scraper for csv documents.

    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    -3
    - Fetch raw csv bytes via the configured client.
    -- Wrap the payload in a canonical `Content` instance with the
    -  CSV content type and source identifier.
    -
    -

    Constraints:

    -
    1
    -2
    - The scraper does not perform parsing or interpretation.
    -- Does not assume a specific storage backend.
    -
    -
    -

    Initialize the CSV scraper.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    client - BaseCsvClient - -
    -

    Client responsible for retrieving raw csv bytes.

    -
    -
    - required -
    - - - - -
    - - - - - - - -
    Functions
    - -
    - - -
    - fetch - - -
    -
    1
    -2
    -3
    -4
    -5
    fetch(
    -    source: Any,
    -    *,
    -    metadata: Mapping[str, Any] | None = None
    -) -> Content
    -
    - -
    - -

    Fetch a csv document from the given source.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    source - Any - -
    -

    Identifier of the csv source as understood by the -configured client.

    -
    -
    - required -
    metadata - Mapping[str, Any] | None - -
    -

    Optional metadata to attach to the returned content.

    -
    -
    - None -
    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    Content - Content - -
    -

    A Content instance containing raw csv bytes, source -identifier, CSV content type, and optional metadata.

    -
    -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - Exception - -
    -

    Retrieval-specific errors raised by the client.

    -
    -
    - -
    - -
    - - - -
    - -
    - -
    - - - - -
    - -
    - -
    - - - - - - - - - - - - - -
    -
    - - - - - -
    - - - -
    - - - -
    -
    -
    -
    - - - - - - - - - - - - \ No newline at end of file diff --git a/omniread/lib/omniread/html/index.html b/omniread/lib/omniread/html/index.html deleted file mode 100644 index 7fa6093..0000000 --- a/omniread/lib/omniread/html/index.html +++ /dev/null @@ -1,1904 +0,0 @@ - - - - - - - - - - - - - - - - - - - Html - omniread - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -
    - -
    - - - - - - -
    - - -
    - -
    - - - - - - -
    -
    - - - -
    -
    -
    - - - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    - - - - - -

    Html

    - - -
    - - - -

    - omniread.html - - -

    - -
    - -

    Summary

    -

    HTML format implementation for OmniRead.

    -

    This package provides HTML-specific implementations of the core OmniRead -contracts defined in omniread.core.

    -

    It includes:

    -
      -
    • HTML parsers that interpret HTML content.
    • -
    • HTML scrapers that retrieve HTML documents.
    • -
    -

    Key characteristics:

    -
      -
    • Implements, but does not redefine, core contracts.
    • -
    • May contain HTML-specific behavior and edge-case handling.
    • -
    • Produces canonical content models defined in omniread.core.content.
    • -
    -

    Consumers should depend on omniread.core interfaces wherever possible and -use this package only when HTML-specific behavior is required.

    -
    -

    Public API

    -
      -
    • HTMLScraper
    • -
    • HTMLParser
    • -
    -
    - - - -
    - - - - - - -

    Classes

    - -
    - - - -

    - HTMLParser - - -

    -
    HTMLParser(content: Content, features: str = 'html.parser')
    -
    - -
    -

    - Bases: BaseParser[T], Generic[T]

    - - -

    Base HTML parser.

    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    -3
    -4
    - This class extends the core `BaseParser` with HTML-specific behavior,
    -  including DOM parsing via BeautifulSoup and reusable extraction helpers.
    -- Provides reusable helpers for HTML extraction. Concrete parsers must
    -  explicitly define the return type.
    -
    -

    Guarantees:

    -
    1
    -2
    -3
    - Accepts only HTML content.
    -- Owns a parsed BeautifulSoup DOM tree.
    -- Provides pure helper utilities for common HTML structures.
    -
    -

    Constraints:

    -
    1
    -2
    - Concrete subclasses must define the output type `T` and implement
    -  the `parse()` method.
    -
    -
    -

    Initialize the HTML parser.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    content - Content - -
    -

    HTML content to be parsed.

    -
    -
    - required -
    features - str - -
    -

    BeautifulSoup parser backend to use (e.g., 'html.parser', 'lxml').

    -
    -
    - 'html.parser' -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - ValueError - -
    -

    If the content is empty or not valid HTML.

    -
    -
    - - - - -
    - - - - - -
    Attributes
    - -
    - - - -
    - supported_types - - - - class-attribute - instance-attribute - - -
    -
    supported_types: set[ContentType] = {HTML}
    -
    - -
    - -

    Set of content types supported by this parser (HTML only).

    -
    - -
    - -
    Functions
    - -
    - - -
    - parse - - - - abstractmethod - - -
    -
    parse() -> T
    -
    - -
    - -

    Fully parse the HTML content into structured output.

    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    T - T - -
    -

    Parsed representation of type T.

    -
    -
    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    - Implementations must fully interpret the HTML DOM and return a
    -  deterministic, structured output.
    -
    -
    -
    - -
    - -
    - - -
    - parse_div - - - - staticmethod - - -
    -
    parse_div(div: Tag, *, separator: str = ' ') -> str
    -
    - -
    - -

    Extract normalized text from a <div> element.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    div - Tag - -
    -

    BeautifulSoup tag representing a <div>.

    -
    -
    - required -
    separator - str - -
    -

    String used to separate text nodes.

    -
    -
    - ' ' -
    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    str - str - -
    -

    Flattened, whitespace-normalized text content.

    -
    -
    - -
    - -
    - -
    - - - -
    parse_link(a: Tag) -> str | None
    -
    - -
    - -

    Extract the hyperlink reference from an <a> element.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    a - Tag - -
    -

    BeautifulSoup tag representing an anchor.

    -
    -
    - required -
    - - -

    Returns:

    - - - - - - - - - - - - - -
    TypeDescription
    - str | None - -
    -

    str | None: -The value of the href attribute, or None if absent.

    -
    -
    - -
    - -
    - -
    - - -
    - parse_meta - - -
    -
    parse_meta() -> dict[str, Any]
    -
    - -
    - -

    Extract high-level metadata from the HTML document.

    - - -

    Returns:

    - - - - - - - - - - - - - -
    TypeDescription
    - dict[str, Any] - -
    -

    dict[str, Any]: -Dictionary containing extracted metadata.

    -
    -
    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    -3
    - Extract high-level metadata from the HTML document.
    -- This includes: Document title, `<meta>` tag name/property to
    -  content mappings.
    -
    -
    -
    - -
    - -
    - - -
    - parse_table - - - - staticmethod - - -
    -
    parse_table(table: Tag) -> list[list[str]]
    -
    - -
    - -

    Parse an HTML table into a 2D list of strings.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    table - Tag - -
    -

    BeautifulSoup tag representing a <table>.

    -
    -
    - required -
    - - -

    Returns:

    - - - - - - - - - - - - - -
    TypeDescription
    - list[list[str]] - -
    -

    list[list[str]]: -A list of rows, where each row is a list of cell text values.

    -
    -
    - -
    - -
    - -
    - - -
    - supports - - -
    -
    supports() -> bool
    -
    - -
    - -

    Check whether this parser supports the content's type.

    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    bool - bool - -
    -

    True if the content type is supported; False otherwise.

    -
    -
    - -
    - -
    - - - -
    - -
    - -
    - -
    - - - -

    - HTMLScraper - - -

    -
    1
    -2
    -3
    -4
    -5
    -6
    -7
    HTMLScraper(
    -    *,
    -    client: httpx.Client | None = None,
    -    timeout: float = 15.0,
    -    headers: Mapping[str, str] | None = None,
    -    follow_redirects: bool = True
    -)
    -
    - -
    -

    - Bases: BaseScraper

    - - -

    Base HTML scraper using httpx.

    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    -3
    -4
    -5
    - This scraper retrieves HTML documents over HTTP(S) and returns
    -  them as raw content wrapped in a `Content` object.
    -- Fetches raw bytes and metadata only.
    -- The scraper uses `httpx.Client` for HTTP requests, enforces an
    -  HTML content type, and preserves HTTP response metadata.
    -
    -

    Constraints:

    -
    1
    -2
    - The scraper does not: Parse HTML, perform retries or backoff,
    -  handle non-HTML responses.
    -
    -
    -

    Initialize the HTML scraper.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    client - Client | None - -
    -

    Optional pre-configured httpx.Client. If omitted, a client is created internally.

    -
    -
    - None -
    timeout - float - -
    -

    Request timeout in seconds.

    -
    -
    - 15.0 -
    headers - Mapping[str, str] | None - -
    -

    Optional default HTTP headers.

    -
    -
    - None -
    follow_redirects - bool - -
    -

    Whether to follow HTTP redirects.

    -
    -
    - True -
    - - - - -
    - - - - - - - -
    Functions
    - -
    - - -
    - fetch - - -
    -
    1
    -2
    -3
    -4
    -5
    fetch(
    -    source: str,
    -    *,
    -    metadata: Mapping[str, Any] | None = None
    -) -> Content
    -
    - -
    - -

    Fetch an HTML document from the given source.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    source - str - -
    -

    URL of the HTML document.

    -
    -
    - required -
    metadata - Mapping[str, Any] | None - -
    -

    Optional metadata to be merged into the returned content.

    -
    -
    - None -
    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    Content - Content - -
    -

    A Content instance containing raw HTML bytes, source URL, HTML content type, and HTTP response metadata.

    -
    -
    - - -

    Raises:

    - - - - - - - - - - - - - - - - - -
    TypeDescription
    - HTTPError - -
    -

    If the HTTP request fails.

    -
    -
    - ValueError - -
    -

    If the response is not valid HTML.

    -
    -
    - -
    - -
    - -
    - - -
    - validate_content_type - - -
    -
    validate_content_type(response: httpx.Response) -> None
    -
    - -
    - -

    Validate that the HTTP response contains HTML content.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    response - Response - -
    -

    HTTP response returned by httpx.

    -
    -
    - required -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - ValueError - -
    -

    If the Content-Type header is missing or does not indicate HTML content.

    -
    -
    - -
    - -
    - - - -
    - -
    - -
    - - - - -
    - -
    - -
    - - - - - - - - - - - - - -
    -
    - - - - - -
    - - - -
    - - - -
    -
    -
    -
    - - - - - - - - - - - - \ No newline at end of file diff --git a/omniread/lib/omniread/html/parser/index.html b/omniread/lib/omniread/html/parser/index.html deleted file mode 100644 index e272ea8..0000000 --- a/omniread/lib/omniread/html/parser/index.html +++ /dev/null @@ -1,1490 +0,0 @@ - - - - - - - - - - - - - - - - - - - Parser - omniread - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -
    - -
    - - - - - - -
    - - -
    - -
    - - - - - - -
    -
    - - - -
    -
    -
    - - - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    - - - - - -

    Parser

    - - -
    - - - -

    - omniread.html.parser - - -

    - -
    - -

    Summary

    -

    HTML parser base implementations for OmniRead.

    -

    This module provides reusable HTML parsing utilities built on top of -the abstract parser contracts defined in omniread.core.parser.

    -

    It supplies:

    -
      -
    • Content-type enforcement for HTML inputs
    • -
    • BeautifulSoup initialization and lifecycle management
    • -
    • Common helper methods for extracting structured data from HTML elements
    • -
    -

    Concrete parsers must subclass HTMLParser and implement the parse() method -to return a structured representation appropriate for their use case.

    - - - -
    - - - - - - -

    Classes

    - -
    - - - -

    - HTMLParser - - -

    -
    HTMLParser(content: Content, features: str = 'html.parser')
    -
    - -
    -

    - Bases: BaseParser[T], Generic[T]

    - - -

    Base HTML parser.

    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    -3
    -4
    - This class extends the core `BaseParser` with HTML-specific behavior,
    -  including DOM parsing via BeautifulSoup and reusable extraction helpers.
    -- Provides reusable helpers for HTML extraction. Concrete parsers must
    -  explicitly define the return type.
    -
    -

    Guarantees:

    -
    1
    -2
    -3
    - Accepts only HTML content.
    -- Owns a parsed BeautifulSoup DOM tree.
    -- Provides pure helper utilities for common HTML structures.
    -
    -

    Constraints:

    -
    1
    -2
    - Concrete subclasses must define the output type `T` and implement
    -  the `parse()` method.
    -
    -
    -

    Initialize the HTML parser.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    content - Content - -
    -

    HTML content to be parsed.

    -
    -
    - required -
    features - str - -
    -

    BeautifulSoup parser backend to use (e.g., 'html.parser', 'lxml').

    -
    -
    - 'html.parser' -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - ValueError - -
    -

    If the content is empty or not valid HTML.

    -
    -
    - - - - -
    - - - - - -
    Attributes
    - -
    - - - -
    - supported_types - - - - class-attribute - instance-attribute - - -
    -
    supported_types: set[ContentType] = {HTML}
    -
    - -
    - -

    Set of content types supported by this parser (HTML only).

    -
    - -
    - -
    Functions
    - -
    - - -
    - parse - - - - abstractmethod - - -
    -
    parse() -> T
    -
    - -
    - -

    Fully parse the HTML content into structured output.

    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    T - T - -
    -

    Parsed representation of type T.

    -
    -
    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    - Implementations must fully interpret the HTML DOM and return a
    -  deterministic, structured output.
    -
    -
    -
    - -
    - -
    - - -
    - parse_div - - - - staticmethod - - -
    -
    parse_div(div: Tag, *, separator: str = ' ') -> str
    -
    - -
    - -

    Extract normalized text from a <div> element.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    div - Tag - -
    -

    BeautifulSoup tag representing a <div>.

    -
    -
    - required -
    separator - str - -
    -

    String used to separate text nodes.

    -
    -
    - ' ' -
    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    str - str - -
    -

    Flattened, whitespace-normalized text content.

    -
    -
    - -
    - -
    - -
    - - - -
    parse_link(a: Tag) -> str | None
    -
    - -
    - -

    Extract the hyperlink reference from an <a> element.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    a - Tag - -
    -

    BeautifulSoup tag representing an anchor.

    -
    -
    - required -
    - - -

    Returns:

    - - - - - - - - - - - - - -
    TypeDescription
    - str | None - -
    -

    str | None: -The value of the href attribute, or None if absent.

    -
    -
    - -
    - -
    - -
    - - -
    - parse_meta - - -
    -
    parse_meta() -> dict[str, Any]
    -
    - -
    - -

    Extract high-level metadata from the HTML document.

    - - -

    Returns:

    - - - - - - - - - - - - - -
    TypeDescription
    - dict[str, Any] - -
    -

    dict[str, Any]: -Dictionary containing extracted metadata.

    -
    -
    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    -3
    - Extract high-level metadata from the HTML document.
    -- This includes: Document title, `<meta>` tag name/property to
    -  content mappings.
    -
    -
    -
    - -
    - -
    - - -
    - parse_table - - - - staticmethod - - -
    -
    parse_table(table: Tag) -> list[list[str]]
    -
    - -
    - -

    Parse an HTML table into a 2D list of strings.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    table - Tag - -
    -

    BeautifulSoup tag representing a <table>.

    -
    -
    - required -
    - - -

    Returns:

    - - - - - - - - - - - - - -
    TypeDescription
    - list[list[str]] - -
    -

    list[list[str]]: -A list of rows, where each row is a list of cell text values.

    -
    -
    - -
    - -
    - -
    - - -
    - supports - - -
    -
    supports() -> bool
    -
    - -
    - -

    Check whether this parser supports the content's type.

    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    bool - bool - -
    -

    True if the content type is supported; False otherwise.

    -
    -
    - -
    - -
    - - - -
    - -
    - -
    - - - - -
    - -
    - -
    - - - - - - - - - - - - - -
    -
    - - - - - -
    - - - -
    - - - -
    -
    -
    -
    - - - - - - - - - - - - \ No newline at end of file diff --git a/omniread/lib/omniread/html/scraper/index.html b/omniread/lib/omniread/html/scraper/index.html deleted file mode 100644 index 68f4acc..0000000 --- a/omniread/lib/omniread/html/scraper/index.html +++ /dev/null @@ -1,1228 +0,0 @@ - - - - - - - - - - - - - - - - - - - Scraper - omniread - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -
    - -
    - - - - - - -
    - - -
    - -
    - - - - - - -
    -
    - - - -
    -
    -
    - - - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    - - - - - -

    Scraper

    - - -
    - - - -

    - omniread.html.scraper - - -

    - -
    - -

    Summary

    -

    HTML scraping implementation for OmniRead.

    -

    This module provides an HTTP-based scraper for retrieving HTML documents. -It implements the core BaseScraper contract using httpx as the transport -layer.

    -

    This scraper is responsible for:

    -
      -
    • Fetching raw HTML bytes over HTTP(S)
    • -
    • Validating response content type
    • -
    • Attaching HTTP metadata to the returned content
    • -
    -

    This scraper is not responsible for:

    -
      -
    • Parsing or interpreting HTML
    • -
    • Retrying failed requests
    • -
    • Managing crawl policies or rate limiting
    • -
    - - - -
    - - - - - - -

    Classes

    - -
    - - - -

    - HTMLScraper - - -

    -
    1
    -2
    -3
    -4
    -5
    -6
    -7
    HTMLScraper(
    -    *,
    -    client: httpx.Client | None = None,
    -    timeout: float = 15.0,
    -    headers: Mapping[str, str] | None = None,
    -    follow_redirects: bool = True
    -)
    -
    - -
    -

    - Bases: BaseScraper

    - - -

    Base HTML scraper using httpx.

    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    -3
    -4
    -5
    - This scraper retrieves HTML documents over HTTP(S) and returns
    -  them as raw content wrapped in a `Content` object.
    -- Fetches raw bytes and metadata only.
    -- The scraper uses `httpx.Client` for HTTP requests, enforces an
    -  HTML content type, and preserves HTTP response metadata.
    -
    -

    Constraints:

    -
    1
    -2
    - The scraper does not: Parse HTML, perform retries or backoff,
    -  handle non-HTML responses.
    -
    -
    -

    Initialize the HTML scraper.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    client - Client | None - -
    -

    Optional pre-configured httpx.Client. If omitted, a client is created internally.

    -
    -
    - None -
    timeout - float - -
    -

    Request timeout in seconds.

    -
    -
    - 15.0 -
    headers - Mapping[str, str] | None - -
    -

    Optional default HTTP headers.

    -
    -
    - None -
    follow_redirects - bool - -
    -

    Whether to follow HTTP redirects.

    -
    -
    - True -
    - - - - -
    - - - - - - - -
    Functions
    - -
    - - -
    - fetch - - -
    -
    1
    -2
    -3
    -4
    -5
    fetch(
    -    source: str,
    -    *,
    -    metadata: Mapping[str, Any] | None = None
    -) -> Content
    -
    - -
    - -

    Fetch an HTML document from the given source.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    source - str - -
    -

    URL of the HTML document.

    -
    -
    - required -
    metadata - Mapping[str, Any] | None - -
    -

    Optional metadata to be merged into the returned content.

    -
    -
    - None -
    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    Content - Content - -
    -

    A Content instance containing raw HTML bytes, source URL, HTML content type, and HTTP response metadata.

    -
    -
    - - -

    Raises:

    - - - - - - - - - - - - - - - - - -
    TypeDescription
    - HTTPError - -
    -

    If the HTTP request fails.

    -
    -
    - ValueError - -
    -

    If the response is not valid HTML.

    -
    -
    - -
    - -
    - -
    - - -
    - validate_content_type - - -
    -
    validate_content_type(response: httpx.Response) -> None
    -
    - -
    - -

    Validate that the HTTP response contains HTML content.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    response - Response - -
    -

    HTTP response returned by httpx.

    -
    -
    - required -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - ValueError - -
    -

    If the Content-Type header is missing or does not indicate HTML content.

    -
    -
    - -
    - -
    - - - -
    - -
    - -
    - - - - -
    - -
    - -
    - - - - - - - - - - - - - -
    -
    - - - - - -
    - - - -
    - - - -
    -
    -
    -
    - - - - - - - - - - - - \ No newline at end of file diff --git a/omniread/lib/omniread/index.html b/omniread/lib/omniread/index.html deleted file mode 100644 index fd9206b..0000000 --- a/omniread/lib/omniread/index.html +++ /dev/null @@ -1,3294 +0,0 @@ - - - - - - - - - - - - - - - - - - - Omniread - omniread - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -
    - -
    - - - - - - -
    - - -
    - -
    - - - - - - -
    -
    - - - -
    -
    -
    - - - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    - - - - - -

    Omniread

    - - -
    - - - -

    - omniread - - -

    - -
    - -

    Summary

    -

    OmniRead — format-agnostic content acquisition and parsing framework.

    -

    OmniRead provides a cleanly layered architecture for fetching, parsing, -and normalizing content from heterogeneous sources such as HTML documents -and PDF files.

    -

    The library is structured around three core concepts:

    -
      -
    1. Content: A canonical, format-agnostic container representing raw content - bytes and minimal contextual metadata.
    2. -
    3. Scrapers: Components responsible for acquiring raw content from a - source (HTTP, filesystem, object storage, etc.). Scrapers never interpret - content.
    4. -
    5. Parsers: Components responsible for interpreting acquired content and - converting it into structured, typed representations.
    6. -
    -

    OmniRead deliberately separates these responsibilities to ensure:

    -
      -
    • Clear boundaries between IO and interpretation.
    • -
    • Replaceable implementations per format.
    • -
    • Predictable, testable behavior.
    • -
    -

    Installation

    -

    Install OmniRead using pip:

    -
    pip install omniread
    -
    -

    Install OmniRead using Poetry: -

    poetry add omniread
    -

    -
    -

    Quick start

    - - -
    - Example -

    HTML example: -

     1
    - 2
    - 3
    - 4
    - 5
    - 6
    - 7
    - 8
    - 9
    -10
    -11
    from omniread import HTMLScraper, HTMLParser
    -
    -scraper = HTMLScraper()
    -content = scraper.fetch("https://example.com")
    -
    -class TitleParser(HTMLParser[str]):
    -    def parse(self) -> str:
    -        return self._soup.title.string
    -
    -parser = TitleParser(content)
    -title = parser.parse()
    -

    -

    PDF example: -

     1
    - 2
    - 3
    - 4
    - 5
    - 6
    - 7
    - 8
    - 9
    -10
    -11
    -12
    -13
    -14
    from omniread import FileSystemPDFClient, PDFScraper, PDFParser
    -from pathlib import Path
    -
    -client = FileSystemPDFClient()
    -scraper = PDFScraper(client=client)
    -content = scraper.fetch(Path("document.pdf"))
    -
    -class TextPDFParser(PDFParser[str]):
    -    def parse(self) -> str:
    -        # implement PDF text extraction
    -        ...
    -
    -parser = TextPDFParser(content)
    -result = parser.parse()
    -

    -

    -

    Public API

    -

    This module re-exports the recommended public entry points of OmniRead. -Consumers are encouraged to import from this namespace rather than from -format-specific submodules directly, unless advanced customization is -required.

    -
      -
    • Content: Canonical content model.
    • -
    • ContentType: Supported media types.
    • -
    • HTMLScraper: HTTP-based HTML acquisition.
    • -
    • HTMLParser: Base parser for HTML DOM interpretation.
    • -
    • FileSystemPDFClient: Local filesystem PDF access.
    • -
    • PDFScraper: PDF-specific content acquisition.
    • -
    • PDFParser: Base parser for PDF binary interpretation.
    • -
    • FileSystemXlsxClient: Local filesystem spreadsheet access.
    • -
    • XlsxScraper: XLSX-specific content acquisition.
    • -
    • XlsxParser: Generic string-row parser for xlsx workbooks.
    • -
    -
    -

    Core Philosophy

    -

    OmniRead is designed as a decoupled content engine:

    -
      -
    1. Separation of Concerns: Scrapers fetch, Parsers interpret. Neither - knows about the other.
    2. -
    3. Normalized Exchange: All components communicate via the Content model, - ensuring a consistent contract.
    4. -
    5. Format Agnosticism: The core logic is independent of whether the input - is HTML, PDF, or JSON.
    6. -
    -
    - - - -
    - - - - - - -

    Classes

    - -
    - - - -

    - Content - - - - dataclass - - -

    -
    1
    -2
    -3
    -4
    -5
    -6
    Content(
    -    raw: bytes,
    -    source: str,
    -    content_type: ContentType | None = ...,
    -    metadata: Mapping[str, Any] | None = ...,
    -)
    -
    - -
    - - -

    Normalized representation of extracted content.

    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    -3
    -4
    - A `Content` instance represents a raw content payload along with
    -  minimal contextual metadata describing its origin and type.
    -- This class is the primary exchange format between scrapers,
    -  parsers, and downstream consumers.
    -
    -
    - - - -
    - - - - - -
    Attributes
    - -
    - - - -
    - content_type - - - - class-attribute - instance-attribute - - -
    -
    content_type: ContentType | None = None
    -
    - -
    - -

    Optional MIME type of the content, if known.

    -
    - -
    - -
    - - - -
    - metadata - - - - class-attribute - instance-attribute - - -
    -
    metadata: Mapping[str, Any] | None = None
    -
    - -
    - -

    Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes).

    -
    - -
    - -
    - - - -
    - raw - - - - instance-attribute - - -
    -
    raw: bytes
    -
    - -
    - -

    Raw content bytes as retrieved from the source.

    -
    - -
    - -
    - - - -
    - source - - - - instance-attribute - - -
    -
    source: str
    -
    - -
    - -

    Identifier of the content origin (URL, file path, or logical name).

    -
    - -
    - - - - - -
    - -
    - -
    - -
    - - - -

    - ContentType - - -

    - - -
    -

    - Bases: str, Enum

    - - -

    Supported MIME types for extracted content.

    - - -
    - Notes -

    Guarantees:

    -
    1
    -2
    -3
    -4
    - This enum represents the declared or inferred media type of the
    -  content source.
    -- It is primarily used for routing content to the appropriate
    -  parser or downstream consumer.
    -
    -
    - - - -
    - - - - - -
    Attributes
    - -
    - - - -
    - CSV - - - - class-attribute - instance-attribute - - -
    -
    CSV = 'text/csv'
    -
    - -
    - -

    Comma-separated-value document content.

    -
    - -
    - -
    - - - -
    - HTML - - - - class-attribute - instance-attribute - - -
    -
    HTML = 'text/html'
    -
    - -
    - -

    HTML document content.

    -
    - -
    - -
    - - - -
    - JSON - - - - class-attribute - instance-attribute - - -
    -
    JSON = 'application/json'
    -
    - -
    - -

    JSON document content.

    -
    - -
    - -
    - - - -
    - PDF - - - - class-attribute - instance-attribute - - -
    -
    PDF = 'application/pdf'
    -
    - -
    - -

    PDF document content.

    -
    - -
    - -
    - - - -
    - XLSX - - - - class-attribute - instance-attribute - - -
    -
    XLSX = "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet"
    -
    - -
    - -

    Office Open XML spreadsheet (xlsx/xlsm) content.

    -
    - -
    - -
    - - - -
    - XML - - - - class-attribute - instance-attribute - - -
    -
    XML = 'application/xml'
    -
    - -
    - -

    XML document content.

    -
    - -
    - - - - - -
    - -
    - -
    - -
    - - - -

    - FileSystemPDFClient - - -

    - - -
    -

    - Bases: BasePDFClient

    - - -

    PDF client that reads from the local filesystem.

    - - -
    - Notes -

    Guarantees:

    -
    1
    -2
    - This client reads PDF files directly from the disk and returns
    -  their raw binary contents.
    -
    -
    - - - -
    - - - - - - - -
    Functions
    - -
    - - -
    - fetch - - -
    -
    fetch(path: Path) -> bytes
    -
    - -
    - -

    Read a PDF file from the local filesystem.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    path - Path - -
    -

    Filesystem path to the PDF file.

    -
    -
    - required -
    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    bytes - bytes - -
    -

    Raw PDF bytes.

    -
    -
    - - -

    Raises:

    - - - - - - - - - - - - - - - - - -
    TypeDescription
    - FileNotFoundError - -
    -

    If the path does not exist.

    -
    -
    - ValueError - -
    -

    If the path exists but is not a file.

    -
    -
    - -
    - -
    - - - -
    - -
    - -
    - -
    - - - -

    - HTMLParser - - -

    -
    HTMLParser(content: Content, features: str = 'html.parser')
    -
    - -
    -

    - Bases: BaseParser[T], Generic[T]

    - - -

    Base HTML parser.

    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    -3
    -4
    - This class extends the core `BaseParser` with HTML-specific behavior,
    -  including DOM parsing via BeautifulSoup and reusable extraction helpers.
    -- Provides reusable helpers for HTML extraction. Concrete parsers must
    -  explicitly define the return type.
    -
    -

    Guarantees:

    -
    1
    -2
    -3
    - Accepts only HTML content.
    -- Owns a parsed BeautifulSoup DOM tree.
    -- Provides pure helper utilities for common HTML structures.
    -
    -

    Constraints:

    -
    1
    -2
    - Concrete subclasses must define the output type `T` and implement
    -  the `parse()` method.
    -
    -
    -

    Initialize the HTML parser.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    content - Content - -
    -

    HTML content to be parsed.

    -
    -
    - required -
    features - str - -
    -

    BeautifulSoup parser backend to use (e.g., 'html.parser', 'lxml').

    -
    -
    - 'html.parser' -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - ValueError - -
    -

    If the content is empty or not valid HTML.

    -
    -
    - - - - -
    - - - - - -
    Attributes
    - -
    - - - -
    - supported_types - - - - class-attribute - instance-attribute - - -
    -
    supported_types: set[ContentType] = {HTML}
    -
    - -
    - -

    Set of content types supported by this parser (HTML only).

    -
    - -
    - -
    Functions
    - -
    - - -
    - parse - - - - abstractmethod - - -
    -
    parse() -> T
    -
    - -
    - -

    Fully parse the HTML content into structured output.

    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    T - T - -
    -

    Parsed representation of type T.

    -
    -
    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    - Implementations must fully interpret the HTML DOM and return a
    -  deterministic, structured output.
    -
    -
    -
    - -
    - -
    - - -
    - parse_div - - - - staticmethod - - -
    -
    parse_div(div: Tag, *, separator: str = ' ') -> str
    -
    - -
    - -

    Extract normalized text from a <div> element.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    div - Tag - -
    -

    BeautifulSoup tag representing a <div>.

    -
    -
    - required -
    separator - str - -
    -

    String used to separate text nodes.

    -
    -
    - ' ' -
    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    str - str - -
    -

    Flattened, whitespace-normalized text content.

    -
    -
    - -
    - -
    - -
    - - - -
    parse_link(a: Tag) -> str | None
    -
    - -
    - -

    Extract the hyperlink reference from an <a> element.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    a - Tag - -
    -

    BeautifulSoup tag representing an anchor.

    -
    -
    - required -
    - - -

    Returns:

    - - - - - - - - - - - - - -
    TypeDescription
    - str | None - -
    -

    str | None: -The value of the href attribute, or None if absent.

    -
    -
    - -
    - -
    - -
    - - -
    - parse_meta - - -
    -
    parse_meta() -> dict[str, Any]
    -
    - -
    - -

    Extract high-level metadata from the HTML document.

    - - -

    Returns:

    - - - - - - - - - - - - - -
    TypeDescription
    - dict[str, Any] - -
    -

    dict[str, Any]: -Dictionary containing extracted metadata.

    -
    -
    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    -3
    - Extract high-level metadata from the HTML document.
    -- This includes: Document title, `<meta>` tag name/property to
    -  content mappings.
    -
    -
    -
    - -
    - -
    - - -
    - parse_table - - - - staticmethod - - -
    -
    parse_table(table: Tag) -> list[list[str]]
    -
    - -
    - -

    Parse an HTML table into a 2D list of strings.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    table - Tag - -
    -

    BeautifulSoup tag representing a <table>.

    -
    -
    - required -
    - - -

    Returns:

    - - - - - - - - - - - - - -
    TypeDescription
    - list[list[str]] - -
    -

    list[list[str]]: -A list of rows, where each row is a list of cell text values.

    -
    -
    - -
    - -
    - -
    - - -
    - supports - - -
    -
    supports() -> bool
    -
    - -
    - -

    Check whether this parser supports the content's type.

    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    bool - bool - -
    -

    True if the content type is supported; False otherwise.

    -
    -
    - -
    - -
    - - - -
    - -
    - -
    - -
    - - - -

    - HTMLScraper - - -

    -
    1
    -2
    -3
    -4
    -5
    -6
    -7
    HTMLScraper(
    -    *,
    -    client: httpx.Client | None = None,
    -    timeout: float = 15.0,
    -    headers: Mapping[str, str] | None = None,
    -    follow_redirects: bool = True
    -)
    -
    - -
    -

    - Bases: BaseScraper

    - - -

    Base HTML scraper using httpx.

    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    -3
    -4
    -5
    - This scraper retrieves HTML documents over HTTP(S) and returns
    -  them as raw content wrapped in a `Content` object.
    -- Fetches raw bytes and metadata only.
    -- The scraper uses `httpx.Client` for HTTP requests, enforces an
    -  HTML content type, and preserves HTTP response metadata.
    -
    -

    Constraints:

    -
    1
    -2
    - The scraper does not: Parse HTML, perform retries or backoff,
    -  handle non-HTML responses.
    -
    -
    -

    Initialize the HTML scraper.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    client - Client | None - -
    -

    Optional pre-configured httpx.Client. If omitted, a client is created internally.

    -
    -
    - None -
    timeout - float - -
    -

    Request timeout in seconds.

    -
    -
    - 15.0 -
    headers - Mapping[str, str] | None - -
    -

    Optional default HTTP headers.

    -
    -
    - None -
    follow_redirects - bool - -
    -

    Whether to follow HTTP redirects.

    -
    -
    - True -
    - - - - -
    - - - - - - - -
    Functions
    - -
    - - -
    - fetch - - -
    -
    1
    -2
    -3
    -4
    -5
    fetch(
    -    source: str,
    -    *,
    -    metadata: Mapping[str, Any] | None = None
    -) -> Content
    -
    - -
    - -

    Fetch an HTML document from the given source.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    source - str - -
    -

    URL of the HTML document.

    -
    -
    - required -
    metadata - Mapping[str, Any] | None - -
    -

    Optional metadata to be merged into the returned content.

    -
    -
    - None -
    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    Content - Content - -
    -

    A Content instance containing raw HTML bytes, source URL, HTML content type, and HTTP response metadata.

    -
    -
    - - -

    Raises:

    - - - - - - - - - - - - - - - - - -
    TypeDescription
    - HTTPError - -
    -

    If the HTTP request fails.

    -
    -
    - ValueError - -
    -

    If the response is not valid HTML.

    -
    -
    - -
    - -
    - -
    - - -
    - validate_content_type - - -
    -
    validate_content_type(response: httpx.Response) -> None
    -
    - -
    - -

    Validate that the HTTP response contains HTML content.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    response - Response - -
    -

    HTTP response returned by httpx.

    -
    -
    - required -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - ValueError - -
    -

    If the Content-Type header is missing or does not indicate HTML content.

    -
    -
    - -
    - -
    - - - -
    - -
    - -
    - -
    - - - -

    - PDFParser - - -

    -
    PDFParser(content: Content)
    -
    - -
    -

    - Bases: BaseParser[T], Generic[T]

    - - -

    Base PDF parser.

    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    - This class enforces PDF content-type compatibility and provides
    -  the extension point for implementing concrete PDF parsing strategies.
    -
    -

    Constraints:

    -
    1
    -2
    - Concrete implementations must define the output type `T` and
    -  implement the `parse()` method.
    -
    -
    -

    Initialize the parser with content to be parsed.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    content - Content - -
    -

    Content instance to be parsed.

    -
    -
    - required -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - ValueError - -
    -

    If the content type is not supported by this parser.

    -
    -
    - - - - -
    - - - - - -
    Attributes
    - -
    - - - -
    - supported_types - - - - class-attribute - instance-attribute - - -
    -
    supported_types: set[ContentType] = {PDF}
    -
    - -
    - -

    Set of content types supported by this parser (PDF only).

    -
    - -
    - -
    Functions
    - -
    - - -
    - parse - - - - abstractmethod - - -
    -
    parse() -> T
    -
    - -
    - -

    Parse PDF content into a structured output.

    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    T - T - -
    -

    Parsed representation of type T.

    -
    -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - Exception - -
    -

    Parsing-specific errors as defined by the implementation.

    -
    -
    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    - Implementations must fully interpret the PDF binary payload and
    -  return a deterministic, structured output.
    -
    -
    -
    - -
    - -
    - - -
    - supports - - -
    -
    supports() -> bool
    -
    - -
    - -

    Check whether this parser supports the content's type.

    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    bool - bool - -
    -

    True if the content type is supported; False otherwise.

    -
    -
    - -
    - -
    - - - -
    - -
    - -
    - -
    - - - -

    - PDFScraper - - -

    -
    PDFScraper(*, client: BasePDFClient)
    -
    - -
    -

    - Bases: BaseScraper

    - - -

    Scraper for PDF sources.

    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    -3
    - Delegates byte retrieval to a PDF client and normalizes output
    -  into `Content`.
    -- Preserves caller-provided metadata.
    -
    -

    Constraints:

    -
    1
    -2
    - The scraper does not perform parsing or interpretation.
    -- Does not assume a specific storage backend.
    -
    -
    -

    Initialize the PDF scraper.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    client - BasePDFClient - -
    -

    PDF client responsible for retrieving raw PDF bytes.

    -
    -
    - required -
    - - - - -
    - - - - - - - -
    Functions
    - -
    - - -
    - fetch - - -
    -
    1
    -2
    -3
    -4
    -5
    fetch(
    -    source: Any,
    -    *,
    -    metadata: Mapping[str, Any] | None = None
    -) -> Content
    -
    - -
    - -

    Fetch a PDF document from the given source.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    source - Any - -
    -

    Identifier of the PDF source as understood by the configured PDF client.

    -
    -
    - required -
    metadata - Mapping[str, Any] | None - -
    -

    Optional metadata to attach to the returned content.

    -
    -
    - None -
    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    Content - Content - -
    -

    A Content instance containing raw PDF bytes, source identifier, PDF content type, and optional metadata.

    -
    -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - Exception - -
    -

    Retrieval-specific errors raised by the PDF client.

    -
    -
    - -
    - -
    - - - -
    - -
    - -
    - - - - -
    - -
    - -
    - - - - - - - - - - - - - -
    -
    - - - - - -
    - - - -
    - - - -
    -
    -
    -
    - - - - - - - - - - - - \ No newline at end of file diff --git a/omniread/lib/omniread/pdf/client/index.html b/omniread/lib/omniread/pdf/client/index.html deleted file mode 100644 index 8870978..0000000 --- a/omniread/lib/omniread/pdf/client/index.html +++ /dev/null @@ -1,1215 +0,0 @@ - - - - - - - - - - - - - - - - - - - Client - omniread - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -
    - -
    - - - - - - -
    - - -
    - -
    - - - - - - -
    -
    - - - -
    -
    -
    - - - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    - - - - - -

    Client

    - - -
    - - - -

    - omniread.pdf.client - - -

    - -
    - -

    Summary

    -

    PDF client abstractions for OmniRead.

    -

    This module defines the client layer responsible for retrieving raw PDF -bytes from a concrete backing store.

    -

    Clients provide low-level access to PDF binaries and are intentionally -decoupled from scraping and parsing logic. They do not perform validation, -interpretation, or content extraction.

    -

    Typical backing stores include:

    -
      -
    • Local filesystems
    • -
    • Object storage (S3, GCS, etc.)
    • -
    • Network file systems
    • -
    - - - -
    - - - - - - -

    Classes

    - -
    - - - -

    - BasePDFClient - - -

    - - -
    -

    - Bases: ABC

    - - -

    Abstract client responsible for retrieving PDF bytes.

    -

    Retrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).

    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    -3
    -4
    - Implementations must accept a source identifier appropriate to
    -  the backing store.
    -- Return the full PDF binary payload.
    -- Raise retrieval-specific errors on failure.
    -
    -
    - - - -
    - - - - - - - -
    Functions
    - -
    - - -
    - fetch - - - - abstractmethod - - -
    -
    fetch(source: Any) -> bytes
    -
    - -
    - -

    Fetch raw PDF bytes from the given source.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    source - Any - -
    -

    Identifier of the PDF location, such as a file path, object storage key, or remote reference.

    -
    -
    - required -
    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    bytes - bytes - -
    -

    Raw PDF bytes.

    -
    -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - Exception - -
    -

    Retrieval-specific errors defined by the implementation.

    -
    -
    - -
    - -
    - - - -
    - -
    - -
    - -
    - - - -

    - FileSystemPDFClient - - -

    - - -
    -

    - Bases: BasePDFClient

    - - -

    PDF client that reads from the local filesystem.

    - - -
    - Notes -

    Guarantees:

    -
    1
    -2
    - This client reads PDF files directly from the disk and returns
    -  their raw binary contents.
    -
    -
    - - - -
    - - - - - - - -
    Functions
    - -
    - - -
    - fetch - - -
    -
    fetch(path: Path) -> bytes
    -
    - -
    - -

    Read a PDF file from the local filesystem.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    path - Path - -
    -

    Filesystem path to the PDF file.

    -
    -
    - required -
    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    bytes - bytes - -
    -

    Raw PDF bytes.

    -
    -
    - - -

    Raises:

    - - - - - - - - - - - - - - - - - -
    TypeDescription
    - FileNotFoundError - -
    -

    If the path does not exist.

    -
    -
    - ValueError - -
    -

    If the path exists but is not a file.

    -
    -
    - -
    - -
    - - - -
    - -
    - -
    - - - - -
    - -
    - -
    - - - - - - - - - - - - - -
    -
    - - - - - -
    - - - -
    - - - -
    -
    -
    -
    - - - - - - - - - - - - \ No newline at end of file diff --git a/omniread/lib/omniread/pdf/index.html b/omniread/lib/omniread/pdf/index.html deleted file mode 100644 index a59feb9..0000000 --- a/omniread/lib/omniread/pdf/index.html +++ /dev/null @@ -1,1612 +0,0 @@ - - - - - - - - - - - - - - - - - - - Pdf - omniread - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -
    - -
    - - - - - - -
    - - -
    - -
    - - - - - - -
    -
    - - - -
    -
    -
    - - - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    - - - - - -

    Pdf

    - - -
    - - - -

    - omniread.pdf - - -

    - -
    - -

    Summary

    -

    PDF format implementation for OmniRead.

    -

    This package provides PDF-specific implementations of the core OmniRead -contracts defined in omniread.core.

    -

    Unlike HTML, PDF handling requires an explicit client layer for document -access. This package therefore includes:

    -
      -
    • PDF clients for acquiring raw PDF data.
    • -
    • PDF scrapers that coordinate client access.
    • -
    • PDF parsers that extract structured content from PDF binaries.
    • -
    -

    Public exports from this package represent the supported PDF pipeline -and are safe for consumers to import directly when working with PDFs.

    -
    -

    Public API

    -
      -
    • FileSystemPDFClient
    • -
    • PDFScraper
    • -
    • PDFParser
    • -
    -
    - - - -
    - - - - - - -

    Classes

    - -
    - - - -

    - FileSystemPDFClient - - -

    - - -
    -

    - Bases: BasePDFClient

    - - -

    PDF client that reads from the local filesystem.

    - - -
    - Notes -

    Guarantees:

    -
    1
    -2
    - This client reads PDF files directly from the disk and returns
    -  their raw binary contents.
    -
    -
    - - - -
    - - - - - - - -
    Functions
    - -
    - - -
    - fetch - - -
    -
    fetch(path: Path) -> bytes
    -
    - -
    - -

    Read a PDF file from the local filesystem.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    path - Path - -
    -

    Filesystem path to the PDF file.

    -
    -
    - required -
    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    bytes - bytes - -
    -

    Raw PDF bytes.

    -
    -
    - - -

    Raises:

    - - - - - - - - - - - - - - - - - -
    TypeDescription
    - FileNotFoundError - -
    -

    If the path does not exist.

    -
    -
    - ValueError - -
    -

    If the path exists but is not a file.

    -
    -
    - -
    - -
    - - - -
    - -
    - -
    - -
    - - - -

    - PDFParser - - -

    -
    PDFParser(content: Content)
    -
    - -
    -

    - Bases: BaseParser[T], Generic[T]

    - - -

    Base PDF parser.

    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    - This class enforces PDF content-type compatibility and provides
    -  the extension point for implementing concrete PDF parsing strategies.
    -
    -

    Constraints:

    -
    1
    -2
    - Concrete implementations must define the output type `T` and
    -  implement the `parse()` method.
    -
    -
    -

    Initialize the parser with content to be parsed.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    content - Content - -
    -

    Content instance to be parsed.

    -
    -
    - required -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - ValueError - -
    -

    If the content type is not supported by this parser.

    -
    -
    - - - - -
    - - - - - -
    Attributes
    - -
    - - - -
    - supported_types - - - - class-attribute - instance-attribute - - -
    -
    supported_types: set[ContentType] = {PDF}
    -
    - -
    - -

    Set of content types supported by this parser (PDF only).

    -
    - -
    - -
    Functions
    - -
    - - -
    - parse - - - - abstractmethod - - -
    -
    parse() -> T
    -
    - -
    - -

    Parse PDF content into a structured output.

    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    T - T - -
    -

    Parsed representation of type T.

    -
    -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - Exception - -
    -

    Parsing-specific errors as defined by the implementation.

    -
    -
    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    - Implementations must fully interpret the PDF binary payload and
    -  return a deterministic, structured output.
    -
    -
    -
    - -
    - -
    - - -
    - supports - - -
    -
    supports() -> bool
    -
    - -
    - -

    Check whether this parser supports the content's type.

    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    bool - bool - -
    -

    True if the content type is supported; False otherwise.

    -
    -
    - -
    - -
    - - - -
    - -
    - -
    - -
    - - - -

    - PDFScraper - - -

    -
    PDFScraper(*, client: BasePDFClient)
    -
    - -
    -

    - Bases: BaseScraper

    - - -

    Scraper for PDF sources.

    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    -3
    - Delegates byte retrieval to a PDF client and normalizes output
    -  into `Content`.
    -- Preserves caller-provided metadata.
    -
    -

    Constraints:

    -
    1
    -2
    - The scraper does not perform parsing or interpretation.
    -- Does not assume a specific storage backend.
    -
    -
    -

    Initialize the PDF scraper.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    client - BasePDFClient - -
    -

    PDF client responsible for retrieving raw PDF bytes.

    -
    -
    - required -
    - - - - -
    - - - - - - - -
    Functions
    - -
    - - -
    - fetch - - -
    -
    1
    -2
    -3
    -4
    -5
    fetch(
    -    source: Any,
    -    *,
    -    metadata: Mapping[str, Any] | None = None
    -) -> Content
    -
    - -
    - -

    Fetch a PDF document from the given source.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    source - Any - -
    -

    Identifier of the PDF source as understood by the configured PDF client.

    -
    -
    - required -
    metadata - Mapping[str, Any] | None - -
    -

    Optional metadata to attach to the returned content.

    -
    -
    - None -
    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    Content - Content - -
    -

    A Content instance containing raw PDF bytes, source identifier, PDF content type, and optional metadata.

    -
    -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - Exception - -
    -

    Retrieval-specific errors raised by the PDF client.

    -
    -
    - -
    - -
    - - - -
    - -
    - -
    - - - - -
    - -
    - -
    - - - - - - - - - - - - - -
    -
    - - - - - -
    - - - -
    - - - -
    -
    -
    -
    - - - - - - - - - - - - \ No newline at end of file diff --git a/omniread/lib/omniread/pdf/parser/index.html b/omniread/lib/omniread/pdf/parser/index.html deleted file mode 100644 index 9d9225e..0000000 --- a/omniread/lib/omniread/pdf/parser/index.html +++ /dev/null @@ -1,1151 +0,0 @@ - - - - - - - - - - - - - - - - - - - Parser - omniread - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -
    - -
    - - - - - - -
    - - -
    - -
    - - - - - - -
    -
    - - - -
    -
    -
    - - - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    - - - - - -

    Parser

    - - -
    - - - -

    - omniread.pdf.parser - - -

    - -
    - -

    Summary

    -

    PDF parser base implementations for OmniRead.

    -

    This module defines the PDF-specific parser contract, extending the -format-agnostic BaseParser with constraints appropriate for PDF content.

    -

    PDF parsers are responsible for interpreting binary PDF data and producing -structured representations suitable for downstream consumption.

    - - - -
    - - - - - - -

    Classes

    - -
    - - - -

    - PDFParser - - -

    -
    PDFParser(content: Content)
    -
    - -
    -

    - Bases: BaseParser[T], Generic[T]

    - - -

    Base PDF parser.

    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    - This class enforces PDF content-type compatibility and provides
    -  the extension point for implementing concrete PDF parsing strategies.
    -
    -

    Constraints:

    -
    1
    -2
    - Concrete implementations must define the output type `T` and
    -  implement the `parse()` method.
    -
    -
    -

    Initialize the parser with content to be parsed.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    content - Content - -
    -

    Content instance to be parsed.

    -
    -
    - required -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - ValueError - -
    -

    If the content type is not supported by this parser.

    -
    -
    - - - - -
    - - - - - -
    Attributes
    - -
    - - - -
    - supported_types - - - - class-attribute - instance-attribute - - -
    -
    supported_types: set[ContentType] = {PDF}
    -
    - -
    - -

    Set of content types supported by this parser (PDF only).

    -
    - -
    - -
    Functions
    - -
    - - -
    - parse - - - - abstractmethod - - -
    -
    parse() -> T
    -
    - -
    - -

    Parse PDF content into a structured output.

    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    T - T - -
    -

    Parsed representation of type T.

    -
    -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - Exception - -
    -

    Parsing-specific errors as defined by the implementation.

    -
    -
    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    - Implementations must fully interpret the PDF binary payload and
    -  return a deterministic, structured output.
    -
    -
    -
    - -
    - -
    - - -
    - supports - - -
    -
    supports() -> bool
    -
    - -
    - -

    Check whether this parser supports the content's type.

    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    bool - bool - -
    -

    True if the content type is supported; False otherwise.

    -
    -
    - -
    - -
    - - - -
    - -
    - -
    - - - - -
    - -
    - -
    - - - - - - - - - - - - - -
    -
    - - - - - -
    - - - -
    - - - -
    -
    -
    -
    - - - - - - - - - - - - \ No newline at end of file diff --git a/omniread/lib/omniread/pdf/scraper/index.html b/omniread/lib/omniread/pdf/scraper/index.html deleted file mode 100644 index 3a7fc9f..0000000 --- a/omniread/lib/omniread/pdf/scraper/index.html +++ /dev/null @@ -1,1069 +0,0 @@ - - - - - - - - - - - - - - - - - - - Scraper - omniread - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -
    - -
    - - - - - - -
    - - -
    - -
    - - - - - - -
    -
    - - - -
    -
    -
    - - - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    - - - - - -

    Scraper

    - - -
    - - - -

    - omniread.pdf.scraper - - -

    - -
    - -

    Summary

    -

    PDF scraping implementation for OmniRead.

    -

    This module provides a PDF-specific scraper that coordinates PDF byte -retrieval via a client and normalizes the result into a Content object.

    -

    The scraper implements the core BaseScraper contract while delegating -all storage and access concerns to a BasePDFClient implementation.

    - - - -
    - - - - - - -

    Classes

    - -
    - - - -

    - PDFScraper - - -

    -
    PDFScraper(*, client: BasePDFClient)
    -
    - -
    -

    - Bases: BaseScraper

    - - -

    Scraper for PDF sources.

    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    -3
    - Delegates byte retrieval to a PDF client and normalizes output
    -  into `Content`.
    -- Preserves caller-provided metadata.
    -
    -

    Constraints:

    -
    1
    -2
    - The scraper does not perform parsing or interpretation.
    -- Does not assume a specific storage backend.
    -
    -
    -

    Initialize the PDF scraper.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    client - BasePDFClient - -
    -

    PDF client responsible for retrieving raw PDF bytes.

    -
    -
    - required -
    - - - - -
    - - - - - - - -
    Functions
    - -
    - - -
    - fetch - - -
    -
    1
    -2
    -3
    -4
    -5
    fetch(
    -    source: Any,
    -    *,
    -    metadata: Mapping[str, Any] | None = None
    -) -> Content
    -
    - -
    - -

    Fetch a PDF document from the given source.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    source - Any - -
    -

    Identifier of the PDF source as understood by the configured PDF client.

    -
    -
    - required -
    metadata - Mapping[str, Any] | None - -
    -

    Optional metadata to attach to the returned content.

    -
    -
    - None -
    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    Content - Content - -
    -

    A Content instance containing raw PDF bytes, source identifier, PDF content type, and optional metadata.

    -
    -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - Exception - -
    -

    Retrieval-specific errors raised by the PDF client.

    -
    -
    - -
    - -
    - - - -
    - -
    - -
    - - - - -
    - -
    - -
    - - - - - - - - - - - - - -
    -
    - - - - - -
    - - - -
    - - - -
    -
    -
    -
    - - - - - - - - - - - - \ No newline at end of file diff --git a/omniread/lib/omniread/xlsx/client/index.html b/omniread/lib/omniread/xlsx/client/index.html deleted file mode 100644 index 9865ca4..0000000 --- a/omniread/lib/omniread/xlsx/client/index.html +++ /dev/null @@ -1,1216 +0,0 @@ - - - - - - - - - - - - - - - - - - - Client - omniread - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -
    - -
    - - - - - - -
    - - -
    - -
    - - - - - - -
    -
    - - - -
    -
    -
    - - - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    - - - - - -

    Client

    - - -
    - - - -

    - omniread.xlsx.client - - -

    - -
    - -

    Summary

    -

    XLSX client abstractions for OmniRead.

    -

    This module defines the client layer responsible for retrieving raw -Office Open XML spreadsheet bytes from a concrete backing store.

    -

    Clients provide low-level access to xlsx binaries and are intentionally -decoupled from scraping and parsing logic. They do not perform validation, -interpretation, or content extraction.

    -

    Typical backing stores include:

    -
      -
    • Local filesystems
    • -
    • Object storage (S3, GCS, etc.)
    • -
    • Network file systems
    • -
    - - - -
    - - - - - - -

    Classes

    - -
    - - - -

    - BaseXlsxClient - - -

    - - -
    -

    - Bases: ABC

    - - -

    Abstract client responsible for retrieving spreadsheet bytes.

    -

    Retrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).

    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    -3
    -4
    - Implementations must accept a source identifier appropriate to
    -  the backing store.
    -- Return the full xlsx binary payload.
    -- Raise retrieval-specific errors on failure.
    -
    -
    - - - -
    - - - - - - - -
    Functions
    - -
    - - -
    - fetch - - - - abstractmethod - - -
    -
    fetch(source: Any) -> bytes
    -
    - -
    - -

    Fetch raw xlsx bytes from the given source.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    source - Any - -
    -

    Identifier of the spreadsheet location, such as a file path, -object storage key, or remote reference.

    -
    -
    - required -
    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    bytes - bytes - -
    -

    Raw xlsx bytes.

    -
    -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - Exception - -
    -

    Retrieval-specific errors defined by the implementation.

    -
    -
    - -
    - -
    - - - -
    - -
    - -
    - -
    - - - -

    - FileSystemXlsxClient - - -

    - - -
    -

    - Bases: BaseXlsxClient

    - - -

    XLSX client that reads from the local filesystem.

    - - -
    - Notes -

    Guarantees:

    -
    1
    -2
    - This client reads spreadsheet files directly from the disk and
    -  returns their raw binary contents.
    -
    -
    - - - -
    - - - - - - - -
    Functions
    - -
    - - -
    - fetch - - -
    -
    fetch(path: Path) -> bytes
    -
    - -
    - -

    Read an xlsx file from the local filesystem.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    path - Path - -
    -

    Filesystem path to the spreadsheet file.

    -
    -
    - required -
    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    bytes - bytes - -
    -

    Raw xlsx bytes.

    -
    -
    - - -

    Raises:

    - - - - - - - - - - - - - - - - - -
    TypeDescription
    - FileNotFoundError - -
    -

    If the path does not exist.

    -
    -
    - ValueError - -
    -

    If the path exists but is not a file.

    -
    -
    - -
    - -
    - - - -
    - -
    - -
    - - - - -
    - -
    - -
    - - - - - - - - - - - - - -
    -
    - - - - - -
    - - - -
    - - - -
    -
    -
    -
    - - - - - - - - - - - - \ No newline at end of file diff --git a/omniread/lib/omniread/xlsx/index.html b/omniread/lib/omniread/xlsx/index.html deleted file mode 100644 index b0a9b85..0000000 --- a/omniread/lib/omniread/xlsx/index.html +++ /dev/null @@ -1,2279 +0,0 @@ - - - - - - - - - - - - - - - - - - - Xlsx - omniread - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -
    - -
    - - - - - - -
    - - -
    - -
    - - - - - - -
    -
    - - - -
    -
    -
    - - - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    - - - - - -

    Xlsx

    - - -
    - - - -

    - omniread.xlsx - - -

    - -
    - -

    Summary

    -

    XLSX subpackage for OmniRead.

    -

    Provides acquisition and parsing of Office Open XML spreadsheet (xlsx) -content:

    -
      -
    • BaseXlsxClient: abstract backing-store client for xlsx bytes.
    • -
    • FileSystemXlsxClient: local filesystem implementation.
    • -
    • XlsxScraper: wraps fetched bytes into canonical Content.
    • -
    • XlsxParserBase: content-type-enforcing parser contract.
    • -
    • XlsxParser: generic string-row parser built on openpyxl.
    • -
    - - - -
    - - - - - - -

    Classes

    - -
    - - - -

    - BaseXlsxClient - - -

    - - -
    -

    - Bases: ABC

    - - -

    Abstract client responsible for retrieving spreadsheet bytes.

    -

    Retrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).

    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    -3
    -4
    - Implementations must accept a source identifier appropriate to
    -  the backing store.
    -- Return the full xlsx binary payload.
    -- Raise retrieval-specific errors on failure.
    -
    -
    - - - -
    - - - - - - - -
    Functions
    - -
    - - -
    - fetch - - - - abstractmethod - - -
    -
    fetch(source: Any) -> bytes
    -
    - -
    - -

    Fetch raw xlsx bytes from the given source.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    source - Any - -
    -

    Identifier of the spreadsheet location, such as a file path, -object storage key, or remote reference.

    -
    -
    - required -
    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    bytes - bytes - -
    -

    Raw xlsx bytes.

    -
    -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - Exception - -
    -

    Retrieval-specific errors defined by the implementation.

    -
    -
    - -
    - -
    - - - -
    - -
    - -
    - -
    - - - -

    - FileSystemXlsxClient - - -

    - - -
    -

    - Bases: BaseXlsxClient

    - - -

    XLSX client that reads from the local filesystem.

    - - -
    - Notes -

    Guarantees:

    -
    1
    -2
    - This client reads spreadsheet files directly from the disk and
    -  returns their raw binary contents.
    -
    -
    - - - -
    - - - - - - - -
    Functions
    - -
    - - -
    - fetch - - -
    -
    fetch(path: Path) -> bytes
    -
    - -
    - -

    Read an xlsx file from the local filesystem.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    path - Path - -
    -

    Filesystem path to the spreadsheet file.

    -
    -
    - required -
    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    bytes - bytes - -
    -

    Raw xlsx bytes.

    -
    -
    - - -

    Raises:

    - - - - - - - - - - - - - - - - - -
    TypeDescription
    - FileNotFoundError - -
    -

    If the path does not exist.

    -
    -
    - ValueError - -
    -

    If the path exists but is not a file.

    -
    -
    - -
    - -
    - - - -
    - -
    - -
    - -
    - - - -

    - XlsxParser - - -

    -
    1
    -2
    -3
    -4
    -5
    -6
    XlsxParser(
    -    content: Content,
    -    *,
    -    data_only: bool = True,
    -    read_only: bool = True
    -)
    -
    - -
    -

    - Bases: XlsxParserBase[list[list[str]]]

    - - -

    Generic xlsx parser producing string rows from a worksheet.

    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    -3
    -4
    - Lazily load the workbook owned by the parser's content.
    -- Normalize cells (including dates and numeric values) into
    -  deterministic string representations.
    -- Expose sheet discovery and row extraction helpers.
    -
    -

    Constraints:

    -
    1
    -2
    -3
    - Cells are rendered with ``str(value)`` after trimming; date and
    -  datetime values are rendered in ISO format. Consumers requiring
    -  locale-specific formatting must convert on their side.
    -
    -
    -

    Initialize the parser.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    content - Content - -
    -

    XLSX content to parse; its type must be supported.

    -
    -
    - required -
    data_only - bool - -
    -

    Passed to openpyxl: when True, formula cells yield their last -computed value instead of the formula string.

    -
    -
    - True -
    read_only - bool - -
    -

    Passed to openpyxl: streaming mode for lower memory usage.

    -
    -
    - True -
    - - - - -
    - - - - - -
    Attributes
    - -
    - - - -
    - sheet_names - - - - property - - -
    -
    sheet_names: list[str]
    -
    - -
    - -

    Names of all worksheets contained in the workbook.

    -
    - -
    - -
    - - - -
    - supported_types - - - - class-attribute - instance-attribute - - -
    -
    supported_types: set[ContentType] = {XLSX}
    -
    - -
    - -

    Set of content types supported by this parser (XLSX only).

    -
    - -
    - -
    - - - -
    - workbook - - - - property - - -
    -
    workbook: Workbook
    -
    - -
    - -

    The lazily loaded workbook backing this parser's content.

    -
    - -
    - -
    Functions
    - -
    - - -
    - parse - - -
    -
    parse() -> list[list[str]]
    -
    - -
    - -

    Parse the first worksheet into normalized string rows.

    - - -

    Returns:

    - - - - - - - - - - - - - -
    TypeDescription
    - list[list[str]] - -
    -

    list[list[str]]: -Rows of the default (first) worksheet.

    -
    -
    - -
    - -
    - -
    - - -
    - rows - - -
    -
    1
    -2
    -3
    -4
    -5
    rows(
    -    sheet: int | str | None = None,
    -    *,
    -    skip_empty: bool = True
    -) -> list[list[str]]
    -
    - -
    - -

    Extract normalized string rows from a worksheet.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    sheet - int | str | None - -
    -

    Worksheet index or title; defaults to the first worksheet.

    -
    -
    - None -
    skip_empty - bool - -
    -

    When True (default), rows whose cells are all blank are omitted.

    -
    -
    - True -
    - - -

    Returns:

    - - - - - - - - - - - - - -
    TypeDescription
    - list[list[str]] - -
    -

    list[list[str]]: -Normalized rows; trailing blank cells are trimmed per row.

    -
    -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - ValueError - -
    -

    If the requested sheet does not exist.

    -
    -
    - -
    - -
    - -
    - - -
    - supports - - -
    -
    supports() -> bool
    -
    - -
    - -

    Check whether this parser supports the content's type.

    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    bool - bool - -
    -

    True if the content type is supported; False otherwise.

    -
    -
    - -
    - -
    - - - -
    - -
    - -
    - -
    - - - -

    - XlsxParserBase - - -

    -
    XlsxParserBase(content: Content)
    -
    - -
    -

    - Bases: BaseParser[T], Generic[T]

    - - -

    Base xlsx parser.

    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    -3
    - This class enforces xlsx content-type compatibility and provides
    -  the extension point for implementing concrete xlsx parsing
    -  strategies.
    -
    -

    Constraints:

    -
    1
    -2
    - Concrete implementations must define the output type `T` and
    -  implement the `parse()` method.
    -
    -
    -

    Initialize the parser with content to be parsed.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    content - Content - -
    -

    Content instance to be parsed.

    -
    -
    - required -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - ValueError - -
    -

    If the content type is not supported by this parser.

    -
    -
    - - - - -
    - - - - - -
    Attributes
    - -
    - - - -
    - supported_types - - - - class-attribute - instance-attribute - - -
    -
    supported_types: set[ContentType] = {XLSX}
    -
    - -
    - -

    Set of content types supported by this parser (XLSX only).

    -
    - -
    - -
    Functions
    - -
    - - -
    - parse - - - - abstractmethod - - -
    -
    parse() -> T
    -
    - -
    - -

    Parse xlsx content into a structured output.

    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    T - T - -
    -

    Parsed representation of type T.

    -
    -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - Exception - -
    -

    Parsing-specific errors as defined by the implementation.

    -
    -
    - -
    - -
    - -
    - - -
    - supports - - -
    -
    supports() -> bool
    -
    - -
    - -

    Check whether this parser supports the content's type.

    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    bool - bool - -
    -

    True if the content type is supported; False otherwise.

    -
    -
    - -
    - -
    - - - -
    - -
    - -
    - -
    - - - -

    - XlsxScraper - - -

    -
    XlsxScraper(*, client: BaseXlsxClient)
    -
    - -
    - - -

    Scraper for xlsx spreadsheet documents.

    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    -3
    - Fetch raw xlsx bytes via the configured client.
    -- Wrap the payload in a canonical `Content` instance with the
    -  XLSX content type and source identifier.
    -
    -

    Constraints:

    -
    1
    -2
    - The scraper does not perform parsing or interpretation.
    -- Does not assume a specific storage backend.
    -
    -
    -

    Initialize the XLSX scraper.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    client - BaseXlsxClient - -
    -

    Client responsible for retrieving raw spreadsheet bytes.

    -
    -
    - required -
    - - - - -
    - - - - - - - -
    Functions
    - -
    - - -
    - fetch - - -
    -
    1
    -2
    -3
    -4
    -5
    fetch(
    -    source: Any,
    -    *,
    -    metadata: Mapping[str, Any] | None = None
    -) -> Content
    -
    - -
    - -

    Fetch an xlsx document from the given source.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    source - Any - -
    -

    Identifier of the spreadsheet source as understood by the -configured client.

    -
    -
    - required -
    metadata - Mapping[str, Any] | None - -
    -

    Optional metadata to attach to the returned content.

    -
    -
    - None -
    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    Content - Content - -
    -

    A Content instance containing raw xlsx bytes, source -identifier, XLSX content type, and optional metadata.

    -
    -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - Exception - -
    -

    Retrieval-specific errors raised by the client.

    -
    -
    - -
    - -
    - - - -
    - -
    - -
    - - - - -
    - -
    - -
    - - - - - - - - - - - - - -
    -
    - - - - - -
    - - - -
    - - - -
    -
    -
    -
    - - - - - - - - - - - - \ No newline at end of file diff --git a/omniread/lib/omniread/xlsx/parser/index.html b/omniread/lib/omniread/xlsx/parser/index.html deleted file mode 100644 index cb5a42b..0000000 --- a/omniread/lib/omniread/xlsx/parser/index.html +++ /dev/null @@ -1,1330 +0,0 @@ - - - - - - - - - - - - - - - - - - - Parser - omniread - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -
    - -
    - - - - - - -
    - - -
    - -
    - - - - - - -
    -
    - - - -
    -
    -
    - - - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    - - - - - -

    Parser

    - - -
    - - - -

    - omniread.xlsx.parser - - -

    - -
    - -

    Summary

    -

    XLSX parser implementations for OmniRead.

    -

    This module provides a concrete, generic parser for Office Open XML -spreadsheets. It exposes workbook sheets as lists of string rows so -downstream consumers can interpret tabular content without depending on -openpyxl directly.

    -

    The parser is intentionally statement-agnostic: it performs no header -detection or column interpretation beyond basic cell normalization.

    - - - -
    - - - - - - -

    Classes

    - -
    - - - -

    - XlsxParser - - -

    -
    1
    -2
    -3
    -4
    -5
    -6
    XlsxParser(
    -    content: Content,
    -    *,
    -    data_only: bool = True,
    -    read_only: bool = True
    -)
    -
    - -
    -

    - Bases: XlsxParserBase[list[list[str]]]

    - - -

    Generic xlsx parser producing string rows from a worksheet.

    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    -3
    -4
    - Lazily load the workbook owned by the parser's content.
    -- Normalize cells (including dates and numeric values) into
    -  deterministic string representations.
    -- Expose sheet discovery and row extraction helpers.
    -
    -

    Constraints:

    -
    1
    -2
    -3
    - Cells are rendered with ``str(value)`` after trimming; date and
    -  datetime values are rendered in ISO format. Consumers requiring
    -  locale-specific formatting must convert on their side.
    -
    -
    -

    Initialize the parser.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    content - Content - -
    -

    XLSX content to parse; its type must be supported.

    -
    -
    - required -
    data_only - bool - -
    -

    Passed to openpyxl: when True, formula cells yield their last -computed value instead of the formula string.

    -
    -
    - True -
    read_only - bool - -
    -

    Passed to openpyxl: streaming mode for lower memory usage.

    -
    -
    - True -
    - - - - -
    - - - - - -
    Attributes
    - -
    - - - -
    - sheet_names - - - - property - - -
    -
    sheet_names: list[str]
    -
    - -
    - -

    Names of all worksheets contained in the workbook.

    -
    - -
    - -
    - - - -
    - supported_types - - - - class-attribute - instance-attribute - - -
    -
    supported_types: set[ContentType] = {XLSX}
    -
    - -
    - -

    Set of content types supported by this parser (XLSX only).

    -
    - -
    - -
    - - - -
    - workbook - - - - property - - -
    -
    workbook: Workbook
    -
    - -
    - -

    The lazily loaded workbook backing this parser's content.

    -
    - -
    - -
    Functions
    - -
    - - -
    - parse - - -
    -
    parse() -> list[list[str]]
    -
    - -
    - -

    Parse the first worksheet into normalized string rows.

    - - -

    Returns:

    - - - - - - - - - - - - - -
    TypeDescription
    - list[list[str]] - -
    -

    list[list[str]]: -Rows of the default (first) worksheet.

    -
    -
    - -
    - -
    - -
    - - -
    - rows - - -
    -
    1
    -2
    -3
    -4
    -5
    rows(
    -    sheet: int | str | None = None,
    -    *,
    -    skip_empty: bool = True
    -) -> list[list[str]]
    -
    - -
    - -

    Extract normalized string rows from a worksheet.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    sheet - int | str | None - -
    -

    Worksheet index or title; defaults to the first worksheet.

    -
    -
    - None -
    skip_empty - bool - -
    -

    When True (default), rows whose cells are all blank are omitted.

    -
    -
    - True -
    - - -

    Returns:

    - - - - - - - - - - - - - -
    TypeDescription
    - list[list[str]] - -
    -

    list[list[str]]: -Normalized rows; trailing blank cells are trimmed per row.

    -
    -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - ValueError - -
    -

    If the requested sheet does not exist.

    -
    -
    - -
    - -
    - -
    - - -
    - supports - - -
    -
    supports() -> bool
    -
    - -
    - -

    Check whether this parser supports the content's type.

    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    bool - bool - -
    -

    True if the content type is supported; False otherwise.

    -
    -
    - -
    - -
    - - - -
    - -
    - -
    - - - - -
    - -
    - -
    - - - - - - - - - - - - - -
    -
    - - - - - -
    - - - -
    - - - -
    -
    -
    -
    - - - - - - - - - - - - \ No newline at end of file diff --git a/omniread/lib/omniread/xlsx/parser_base/index.html b/omniread/lib/omniread/xlsx/parser_base/index.html deleted file mode 100644 index 6f966cf..0000000 --- a/omniread/lib/omniread/xlsx/parser_base/index.html +++ /dev/null @@ -1,1143 +0,0 @@ - - - - - - - - - - - - - - - - - - - Parser Base - omniread - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -
    - -
    - - - - - - -
    - - -
    - -
    - - - - - - -
    -
    - - - -
    -
    -
    - - - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    - - - - - -

    Parser Base

    - - -
    - - - -

    - omniread.xlsx.parser_base - - -

    - -
    - -

    Summary

    -

    XLSX parser base implementation for OmniRead.

    -

    This module defines the XLSX-specific parser contract, extending the -format-agnostic BaseParser with constraints appropriate for Office Open -XML spreadsheet content.

    - - - -
    - - - - - - -

    Classes

    - -
    - - - -

    - XlsxParserBase - - -

    -
    XlsxParserBase(content: Content)
    -
    - -
    -

    - Bases: BaseParser[T], Generic[T]

    - - -

    Base xlsx parser.

    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    -3
    - This class enforces xlsx content-type compatibility and provides
    -  the extension point for implementing concrete xlsx parsing
    -  strategies.
    -
    -

    Constraints:

    -
    1
    -2
    - Concrete implementations must define the output type `T` and
    -  implement the `parse()` method.
    -
    -
    -

    Initialize the parser with content to be parsed.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    content - Content - -
    -

    Content instance to be parsed.

    -
    -
    - required -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - ValueError - -
    -

    If the content type is not supported by this parser.

    -
    -
    - - - - -
    - - - - - -
    Attributes
    - -
    - - - -
    - supported_types - - - - class-attribute - instance-attribute - - -
    -
    supported_types: set[ContentType] = {XLSX}
    -
    - -
    - -

    Set of content types supported by this parser (XLSX only).

    -
    - -
    - -
    Functions
    - -
    - - -
    - parse - - - - abstractmethod - - -
    -
    parse() -> T
    -
    - -
    - -

    Parse xlsx content into a structured output.

    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    T - T - -
    -

    Parsed representation of type T.

    -
    -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - Exception - -
    -

    Parsing-specific errors as defined by the implementation.

    -
    -
    - -
    - -
    - -
    - - -
    - supports - - -
    -
    supports() -> bool
    -
    - -
    - -

    Check whether this parser supports the content's type.

    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    bool - bool - -
    -

    True if the content type is supported; False otherwise.

    -
    -
    - -
    - -
    - - - -
    - -
    - -
    - - - - -
    - -
    - -
    - - - - - - - - - - - - - -
    -
    - - - - - -
    - - - -
    - - - -
    -
    -
    -
    - - - - - - - - - - - - \ No newline at end of file diff --git a/omniread/lib/omniread/xlsx/scraper/index.html b/omniread/lib/omniread/xlsx/scraper/index.html deleted file mode 100644 index 21c979b..0000000 --- a/omniread/lib/omniread/xlsx/scraper/index.html +++ /dev/null @@ -1,1069 +0,0 @@ - - - - - - - - - - - - - - - - - - - Scraper - omniread - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -
    - -
    - - - - - - -
    - - -
    - -
    - - - - - - -
    -
    - - - -
    -
    -
    - - - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    -
    - - - -
    -
    - - - - - -

    Scraper

    - - -
    - - - -

    - omniread.xlsx.scraper - - -

    - -
    - -

    Summary

    -

    XLSX scraper for OmniRead.

    -

    This module defines the scraper responsible for acquiring raw Office Open -XML spreadsheet content from a backing store via a configured client.

    -

    The scraper does not interpret or parse the acquired bytes; it wraps them in -the canonical Content model.

    - - - -
    - - - - - - -

    Classes

    - -
    - - - -

    - XlsxScraper - - -

    -
    XlsxScraper(*, client: BaseXlsxClient)
    -
    - -
    - - -

    Scraper for xlsx spreadsheet documents.

    - - -
    - Notes -

    Responsibilities:

    -
    1
    -2
    -3
    - Fetch raw xlsx bytes via the configured client.
    -- Wrap the payload in a canonical `Content` instance with the
    -  XLSX content type and source identifier.
    -
    -

    Constraints:

    -
    1
    -2
    - The scraper does not perform parsing or interpretation.
    -- Does not assume a specific storage backend.
    -
    -
    -

    Initialize the XLSX scraper.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    client - BaseXlsxClient - -
    -

    Client responsible for retrieving raw spreadsheet bytes.

    -
    -
    - required -
    - - - - -
    - - - - - - - -
    Functions
    - -
    - - -
    - fetch - - -
    -
    1
    -2
    -3
    -4
    -5
    fetch(
    -    source: Any,
    -    *,
    -    metadata: Mapping[str, Any] | None = None
    -) -> Content
    -
    - -
    - -

    Fetch an xlsx document from the given source.

    - - -

    Parameters:

    - - - - - - - - - - - - - - - - - - - - - - - -
    NameTypeDescriptionDefault
    source - Any - -
    -

    Identifier of the spreadsheet source as understood by the -configured client.

    -
    -
    - required -
    metadata - Mapping[str, Any] | None - -
    -

    Optional metadata to attach to the returned content.

    -
    -
    - None -
    - - -

    Returns:

    - - - - - - - - - - - - - -
    Name TypeDescription
    Content - Content - -
    -

    A Content instance containing raw xlsx bytes, source -identifier, XLSX content type, and optional metadata.

    -
    -
    - - -

    Raises:

    - - - - - - - - - - - - - -
    TypeDescription
    - Exception - -
    -

    Retrieval-specific errors raised by the client.

    -
    -
    - -
    - -
    - - - -
    - -
    - -
    - - - - -
    - -
    - -
    - - - - - - - - - - - - - -
    -
    - - - - - -
    - - - -
    - - - -
    -
    -
    -
    - - - - - - - - - - - - \ No newline at end of file diff --git a/omniread/lib/pdf/client/index.html b/omniread/lib/pdf/client/index.html index 6315b2b..00431a8 100644 --- a/omniread/lib/pdf/client/index.html +++ b/omniread/lib/pdf/client/index.html @@ -802,6 +802,340 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + + + +
  • + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + + + +
  • + + +
    diff --git a/omniread/lib/pdf/index.html b/omniread/lib/pdf/index.html index 4de1f7d..76a2693 100644 --- a/omniread/lib/pdf/index.html +++ b/omniread/lib/pdf/index.html @@ -643,6 +643,340 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + + + +
  • + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + + + +
  • + + + @@ -1097,7 +1431,7 @@ and are safe for consumers to import directly when working with PDFs.

    - Bases: BaseParser[T], Generic[T]

    + Bases: BaseParser[T], Generic[T]

    Base PDF parser.

    @@ -1133,7 +1467,7 @@ and are safe for consumers to import directly when working with PDFs.

    content - Content + Content
    @@ -1349,7 +1683,7 @@ and are safe for consumers to import directly when working with PDFs.

    - Bases: BaseScraper

    + Bases: BaseScraper

    Scraper for PDF sources.

    @@ -1492,7 +1826,7 @@ and are safe for consumers to import directly when working with PDFs.

    Content - Content + Content
    diff --git a/omniread/lib/pdf/parser/index.html b/omniread/lib/pdf/parser/index.html index 7601915..eb044cf 100644 --- a/omniread/lib/pdf/parser/index.html +++ b/omniread/lib/pdf/parser/index.html @@ -796,6 +796,340 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + + + +
  • + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + + + +
  • + + +
    @@ -996,7 +1330,7 @@ structured representations suitable for downstream consumption.

    - Bases: BaseParser[T], Generic[T]

    + Bases: BaseParser[T], Generic[T]

    Base PDF parser.

    @@ -1032,7 +1366,7 @@ structured representations suitable for downstream consumption.

    content - Content + Content
    diff --git a/omniread/lib/pdf/scraper/index.html b/omniread/lib/pdf/scraper/index.html index df437e2..20fe37a 100644 --- a/omniread/lib/pdf/scraper/index.html +++ b/omniread/lib/pdf/scraper/index.html @@ -12,6 +12,8 @@ + + @@ -761,6 +763,340 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + + + +
  • + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + + + +
  • + + +
    @@ -928,7 +1264,7 @@ all storage and access concerns to a BasePDFClient implementation.<

    - Bases: BaseScraper

    + Bases: BaseScraper

    Scraper for PDF sources.

    @@ -1071,7 +1407,7 @@ all storage and access concerns to a BasePDFClient implementation.<
    Content - Content + Content
    diff --git a/omniread/lib/search/search_index.json b/omniread/lib/search/search_index.json index 2f238d3..15cc302 100644 --- a/omniread/lib/search/search_index.json +++ b/omniread/lib/search/search_index.json @@ -1 +1 @@ -{"config":{"lang":["en"],"separator":"[\\s\\-]+","pipeline":["stopWordFilter"]},"docs":[{"location":"","title":"omniread","text":""},{"location":"#omniread","title":"omniread","text":""},{"location":"#omniread--summary","title":"Summary","text":"

    OmniRead \u2014 format-agnostic content acquisition and parsing framework.

    OmniRead provides a cleanly layered architecture for fetching, parsing, and normalizing content from heterogeneous sources such as HTML documents and PDF files.

    The library is structured around three core concepts:

    1. Content: A canonical, format-agnostic container representing raw content bytes and minimal contextual metadata.
    2. Scrapers: Components responsible for acquiring raw content from a source (HTTP, filesystem, object storage, etc.). Scrapers never interpret content.
    3. Parsers: Components responsible for interpreting acquired content and converting it into structured, typed representations.

    OmniRead deliberately separates these responsibilities to ensure:

    • Clear boundaries between IO and interpretation.
    • Replaceable implementations per format.
    • Predictable, testable behavior.
    "},{"location":"#omniread--installation","title":"Installation","text":"

    Install OmniRead using pip:

    pip install omniread\n

    Install OmniRead using Poetry:

    poetry add omniread\n

    "},{"location":"#omniread--quick-start","title":"Quick start","text":"Example

    HTML example:

    from omniread import HTMLScraper, HTMLParser\n\nscraper = HTMLScraper()\ncontent = scraper.fetch(\"https://example.com\")\n\nclass TitleParser(HTMLParser[str]):\n    def parse(self) -> str:\n        return self._soup.title.string\n\nparser = TitleParser(content)\ntitle = parser.parse()\n

    PDF example:

    from omniread import FileSystemPDFClient, PDFScraper, PDFParser\nfrom pathlib import Path\n\nclient = FileSystemPDFClient()\nscraper = PDFScraper(client=client)\ncontent = scraper.fetch(Path(\"document.pdf\"))\n\nclass TextPDFParser(PDFParser[str]):\n    def parse(self) -> str:\n        # implement PDF text extraction\n        ...\n\nparser = TextPDFParser(content)\nresult = parser.parse()\n

    "},{"location":"#omniread--public-api","title":"Public API","text":"

    This module re-exports the recommended public entry points of OmniRead. Consumers are encouraged to import from this namespace rather than from format-specific submodules directly, unless advanced customization is required.

    • Content: Canonical content model.
    • ContentType: Supported media types.
    • HTMLScraper: HTTP-based HTML acquisition.
    • HTMLParser: Base parser for HTML DOM interpretation.
    • FileSystemPDFClient: Local filesystem PDF access.
    • PDFScraper: PDF-specific content acquisition.
    • PDFParser: Base parser for PDF binary interpretation.
    • FileSystemXlsxClient: Local filesystem spreadsheet access.
    • XlsxScraper: XLSX-specific content acquisition.
    • XlsxParser: Generic string-row parser for xlsx workbooks.
    "},{"location":"#omniread--core-philosophy","title":"Core Philosophy","text":"

    OmniRead is designed as a decoupled content engine:

    1. Separation of Concerns: Scrapers fetch, Parsers interpret. Neither knows about the other.
    2. Normalized Exchange: All components communicate via the Content model, ensuring a consistent contract.
    3. Format Agnosticism: The core logic is independent of whether the input is HTML, PDF, or JSON.
    "},{"location":"#omniread-classes","title":"Classes","text":""},{"location":"#omniread.Content","title":"Content dataclass","text":"
    Content(\n    raw: bytes,\n    source: str,\n    content_type: ContentType | None = ...,\n    metadata: Mapping[str, Any] | None = ...,\n)\n

    Normalized representation of extracted content.

    Notes

    Responsibilities:

    - A `Content` instance represents a raw content payload along with\n  minimal contextual metadata describing its origin and type.\n- This class is the primary exchange format between scrapers,\n  parsers, and downstream consumers.\n
    "},{"location":"#omniread.Content-attributes","title":"Attributes","text":""},{"location":"#omniread.Content.content_type","title":"content_type class-attribute instance-attribute","text":"
    content_type: ContentType | None = None\n

    Optional MIME type of the content, if known.

    "},{"location":"#omniread.Content.metadata","title":"metadata class-attribute instance-attribute","text":"
    metadata: Mapping[str, Any] | None = None\n

    Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes).

    "},{"location":"#omniread.Content.raw","title":"raw instance-attribute","text":"
    raw: bytes\n

    Raw content bytes as retrieved from the source.

    "},{"location":"#omniread.Content.source","title":"source instance-attribute","text":"
    source: str\n

    Identifier of the content origin (URL, file path, or logical name).

    "},{"location":"#omniread.ContentType","title":"ContentType","text":"

    Bases: str, Enum

    Supported MIME types for extracted content.

    Notes

    Guarantees:

    - This enum represents the declared or inferred media type of the\n  content source.\n- It is primarily used for routing content to the appropriate\n  parser or downstream consumer.\n
    "},{"location":"#omniread.ContentType-attributes","title":"Attributes","text":""},{"location":"#omniread.ContentType.CSV","title":"CSV class-attribute instance-attribute","text":"
    CSV = 'text/csv'\n

    Comma-separated-value document content.

    "},{"location":"#omniread.ContentType.HTML","title":"HTML class-attribute instance-attribute","text":"
    HTML = 'text/html'\n

    HTML document content.

    "},{"location":"#omniread.ContentType.JSON","title":"JSON class-attribute instance-attribute","text":"
    JSON = 'application/json'\n

    JSON document content.

    "},{"location":"#omniread.ContentType.PDF","title":"PDF class-attribute instance-attribute","text":"
    PDF = 'application/pdf'\n

    PDF document content.

    "},{"location":"#omniread.ContentType.XLSX","title":"XLSX class-attribute instance-attribute","text":"
    XLSX = \"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet\"\n

    Office Open XML spreadsheet (xlsx/xlsm) content.

    "},{"location":"#omniread.ContentType.XML","title":"XML class-attribute instance-attribute","text":"
    XML = 'application/xml'\n

    XML document content.

    "},{"location":"#omniread.FileSystemPDFClient","title":"FileSystemPDFClient","text":"

    Bases: BasePDFClient

    PDF client that reads from the local filesystem.

    Notes

    Guarantees:

    - This client reads PDF files directly from the disk and returns\n  their raw binary contents.\n
    "},{"location":"#omniread.FileSystemPDFClient-functions","title":"Functions","text":""},{"location":"#omniread.FileSystemPDFClient.fetch","title":"fetch","text":"
    fetch(path: Path) -> bytes\n

    Read a PDF file from the local filesystem.

    Parameters:

    Name Type Description Default path Path

    Filesystem path to the PDF file.

    required

    Returns:

    Name Type Description bytes bytes

    Raw PDF bytes.

    Raises:

    Type Description FileNotFoundError

    If the path does not exist.

    ValueError

    If the path exists but is not a file.

    "},{"location":"#omniread.HTMLParser","title":"HTMLParser","text":"
    HTMLParser(content: Content, features: str = 'html.parser')\n

    Bases: BaseParser[T], Generic[T]

    Base HTML parser.

    Notes

    Responsibilities:

    - This class extends the core `BaseParser` with HTML-specific behavior,\n  including DOM parsing via BeautifulSoup and reusable extraction helpers.\n- Provides reusable helpers for HTML extraction. Concrete parsers must\n  explicitly define the return type.\n

    Guarantees:

    - Accepts only HTML content.\n- Owns a parsed BeautifulSoup DOM tree.\n- Provides pure helper utilities for common HTML structures.\n

    Constraints:

    - Concrete subclasses must define the output type `T` and implement\n  the `parse()` method.\n

    Initialize the HTML parser.

    Parameters:

    Name Type Description Default content Content

    HTML content to be parsed.

    required features str

    BeautifulSoup parser backend to use (e.g., 'html.parser', 'lxml').

    'html.parser'

    Raises:

    Type Description ValueError

    If the content is empty or not valid HTML.

    "},{"location":"#omniread.HTMLParser-attributes","title":"Attributes","text":""},{"location":"#omniread.HTMLParser.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {HTML}\n

    Set of content types supported by this parser (HTML only).

    "},{"location":"#omniread.HTMLParser-functions","title":"Functions","text":""},{"location":"#omniread.HTMLParser.parse","title":"parse abstractmethod","text":"
    parse() -> T\n

    Fully parse the HTML content into structured output.

    Returns:

    Name Type Description T T

    Parsed representation of type T.

    Notes

    Responsibilities:

    - Implementations must fully interpret the HTML DOM and return a\n  deterministic, structured output.\n
    "},{"location":"#omniread.HTMLParser.parse_div","title":"parse_div staticmethod","text":"
    parse_div(div: Tag, *, separator: str = ' ') -> str\n

    Extract normalized text from a <div> element.

    Parameters:

    Name Type Description Default div Tag

    BeautifulSoup tag representing a <div>.

    required separator str

    String used to separate text nodes.

    ' '

    Returns:

    Name Type Description str str

    Flattened, whitespace-normalized text content.

    "},{"location":"#omniread.HTMLParser.parse_link","title":"parse_link staticmethod","text":"
    parse_link(a: Tag) -> str | None\n

    Extract the hyperlink reference from an <a> element.

    Parameters:

    Name Type Description Default a Tag

    BeautifulSoup tag representing an anchor.

    required

    Returns:

    Type Description str | None

    str | None: The value of the href attribute, or None if absent.

    "},{"location":"#omniread.HTMLParser.parse_meta","title":"parse_meta","text":"
    parse_meta() -> dict[str, Any]\n

    Extract high-level metadata from the HTML document.

    Returns:

    Type Description dict[str, Any]

    dict[str, Any]: Dictionary containing extracted metadata.

    Notes

    Responsibilities:

    - Extract high-level metadata from the HTML document.\n- This includes: Document title, `<meta>` tag name/property to\n  content mappings.\n
    "},{"location":"#omniread.HTMLParser.parse_table","title":"parse_table staticmethod","text":"
    parse_table(table: Tag) -> list[list[str]]\n

    Parse an HTML table into a 2D list of strings.

    Parameters:

    Name Type Description Default table Tag

    BeautifulSoup tag representing a <table>.

    required

    Returns:

    Type Description list[list[str]]

    list[list[str]]: A list of rows, where each row is a list of cell text values.

    "},{"location":"#omniread.HTMLParser.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"#omniread.HTMLScraper","title":"HTMLScraper","text":"
    HTMLScraper(\n    *,\n    client: httpx.Client | None = None,\n    timeout: float = 15.0,\n    headers: Mapping[str, str] | None = None,\n    follow_redirects: bool = True\n)\n

    Bases: BaseScraper

    Base HTML scraper using httpx.

    Notes

    Responsibilities:

    - This scraper retrieves HTML documents over HTTP(S) and returns\n  them as raw content wrapped in a `Content` object.\n- Fetches raw bytes and metadata only.\n- The scraper uses `httpx.Client` for HTTP requests, enforces an\n  HTML content type, and preserves HTTP response metadata.\n

    Constraints:

    - The scraper does not: Parse HTML, perform retries or backoff,\n  handle non-HTML responses.\n

    Initialize the HTML scraper.

    Parameters:

    Name Type Description Default client Client | None

    Optional pre-configured httpx.Client. If omitted, a client is created internally.

    None timeout float

    Request timeout in seconds.

    15.0 headers Mapping[str, str] | None

    Optional default HTTP headers.

    None follow_redirects bool

    Whether to follow HTTP redirects.

    True"},{"location":"#omniread.HTMLScraper-functions","title":"Functions","text":""},{"location":"#omniread.HTMLScraper.fetch","title":"fetch","text":"
    fetch(\n    source: str,\n    *,\n    metadata: Mapping[str, Any] | None = None\n) -> Content\n

    Fetch an HTML document from the given source.

    Parameters:

    Name Type Description Default source str

    URL of the HTML document.

    required metadata Mapping[str, Any] | None

    Optional metadata to be merged into the returned content.

    None

    Returns:

    Name Type Description Content Content

    A Content instance containing raw HTML bytes, source URL, HTML content type, and HTTP response metadata.

    Raises:

    Type Description HTTPError

    If the HTTP request fails.

    ValueError

    If the response is not valid HTML.

    "},{"location":"#omniread.HTMLScraper.validate_content_type","title":"validate_content_type","text":"
    validate_content_type(response: httpx.Response) -> None\n

    Validate that the HTTP response contains HTML content.

    Parameters:

    Name Type Description Default response Response

    HTTP response returned by httpx.

    required

    Raises:

    Type Description ValueError

    If the Content-Type header is missing or does not indicate HTML content.

    "},{"location":"#omniread.PDFParser","title":"PDFParser","text":"
    PDFParser(content: Content)\n

    Bases: BaseParser[T], Generic[T]

    Base PDF parser.

    Notes

    Responsibilities:

    - This class enforces PDF content-type compatibility and provides\n  the extension point for implementing concrete PDF parsing strategies.\n

    Constraints:

    - Concrete implementations must define the output type `T` and\n  implement the `parse()` method.\n

    Initialize the parser with content to be parsed.

    Parameters:

    Name Type Description Default content Content

    Content instance to be parsed.

    required

    Raises:

    Type Description ValueError

    If the content type is not supported by this parser.

    "},{"location":"#omniread.PDFParser-attributes","title":"Attributes","text":""},{"location":"#omniread.PDFParser.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {PDF}\n

    Set of content types supported by this parser (PDF only).

    "},{"location":"#omniread.PDFParser-functions","title":"Functions","text":""},{"location":"#omniread.PDFParser.parse","title":"parse abstractmethod","text":"
    parse() -> T\n

    Parse PDF content into a structured output.

    Returns:

    Name Type Description T T

    Parsed representation of type T.

    Raises:

    Type Description Exception

    Parsing-specific errors as defined by the implementation.

    Notes

    Responsibilities:

    - Implementations must fully interpret the PDF binary payload and\n  return a deterministic, structured output.\n
    "},{"location":"#omniread.PDFParser.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"#omniread.PDFScraper","title":"PDFScraper","text":"
    PDFScraper(*, client: BasePDFClient)\n

    Bases: BaseScraper

    Scraper for PDF sources.

    Notes

    Responsibilities:

    - Delegates byte retrieval to a PDF client and normalizes output\n  into `Content`.\n- Preserves caller-provided metadata.\n

    Constraints:

    - The scraper does not perform parsing or interpretation.\n- Does not assume a specific storage backend.\n

    Initialize the PDF scraper.

    Parameters:

    Name Type Description Default client BasePDFClient

    PDF client responsible for retrieving raw PDF bytes.

    required"},{"location":"#omniread.PDFScraper-functions","title":"Functions","text":""},{"location":"#omniread.PDFScraper.fetch","title":"fetch","text":"
    fetch(\n    source: Any,\n    *,\n    metadata: Mapping[str, Any] | None = None\n) -> Content\n

    Fetch a PDF document from the given source.

    Parameters:

    Name Type Description Default source Any

    Identifier of the PDF source as understood by the configured PDF client.

    required metadata Mapping[str, Any] | None

    Optional metadata to attach to the returned content.

    None

    Returns:

    Name Type Description Content Content

    A Content instance containing raw PDF bytes, source identifier, PDF content type, and optional metadata.

    Raises:

    Type Description Exception

    Retrieval-specific errors raised by the PDF client.

    "},{"location":"core/","title":"Core","text":""},{"location":"core/#omniread.core","title":"omniread.core","text":""},{"location":"core/#omniread.core--summary","title":"Summary","text":"

    Core domain contracts for OmniRead.

    This package defines the format-agnostic domain layer of OmniRead. It exposes canonical content models and abstract interfaces that are implemented by format-specific modules (HTML, PDF, etc.).

    Public exports from this package are considered stable contracts and are safe for downstream consumers to depend on.

    Submodules:

    • content: Canonical content models and enums.
    • parser: Abstract parsing contracts.
    • scraper: Abstract scraping contracts.

    Format-specific behavior must not be introduced at this layer.

    "},{"location":"core/#omniread.core--public-api","title":"Public API","text":"
    • Content
    • ContentType
    "},{"location":"core/#omniread.core-classes","title":"Classes","text":""},{"location":"core/#omniread.core.BaseParser","title":"BaseParser","text":"
    BaseParser(content: Content)\n

    Bases: ABC, Generic[T]

    Base interface for all parsers.

    Notes

    Guarantees:

    - A parser is a self-contained object that owns the `Content` it is\n  responsible for interpreting.\n- Consumers may rely on early validation of content compatibility\n  and type-stable return values from `parse()`.\n

    Responsibilities:

    - Implementations must declare supported content types via `supported_types`.\n- Implementations must raise parsing-specific exceptions from `parse()`.\n- Implementations must remain deterministic for a given input.\n

    Initialize the parser with content to be parsed.

    Parameters:

    Name Type Description Default content Content

    Content instance to be parsed.

    required

    Raises:

    Type Description ValueError

    If the content type is not supported by this parser.

    "},{"location":"core/#omniread.core.BaseParser-attributes","title":"Attributes","text":""},{"location":"core/#omniread.core.BaseParser.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = set()\n

    Set of content types supported by this parser. An empty set indicates that the parser is content-type agnostic.

    "},{"location":"core/#omniread.core.BaseParser-functions","title":"Functions","text":""},{"location":"core/#omniread.core.BaseParser.parse","title":"parse abstractmethod","text":"
    parse() -> T\n

    Parse the owned content into structured output.

    Returns:

    Name Type Description T T

    Parsed, structured representation.

    Raises:

    Type Description Exception

    Parsing-specific errors as defined by the implementation.

    Notes

    Responsibilities:

    - Implementations must fully consume the provided content and\n  return a deterministic, structured output.\n
    "},{"location":"core/#omniread.core.BaseParser.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"core/#omniread.core.BaseScraper","title":"BaseScraper","text":"

    Bases: ABC

    Base interface for all scrapers.

    Notes

    Responsibilities:

    - A scraper is responsible ONLY for fetching raw content (bytes)\n  from a source. It must not interpret or parse it.\n- A scraper is a stateless acquisition component that retrieves raw\n  content from a source and returns it as a `Content` object.\n- Scrapers define how content is obtained, not what the content means.\n- Implementations may vary in transport mechanism, authentication\n  strategy, retry and backoff behavior.\n

    Constraints:

    - Implementations must not parse content, modify content semantics,\n  or couple scraping logic to a specific parser.\n
    "},{"location":"core/#omniread.core.BaseScraper-functions","title":"Functions","text":""},{"location":"core/#omniread.core.BaseScraper.fetch","title":"fetch abstractmethod","text":"
    fetch(\n    source: str,\n    *,\n    metadata: Mapping[str, Any] | None = None\n) -> Content\n

    Fetch raw content from the given source.

    Parameters:

    Name Type Description Default source str

    Location identifier (URL, file path, S3 URI, etc.).

    required metadata Mapping[str, Any] | None

    Optional hints for the scraper (headers, auth, etc.).

    None

    Returns:

    Name Type Description Content Content

    Content object containing raw bytes and metadata.

    Raises:

    Type Description Exception

    Retrieval-specific errors as defined by the implementation.

    Notes

    Responsibilities:

    - Implementations must retrieve the content referenced by `source`\n  and return it as raw bytes wrapped in a `Content` object.\n
    "},{"location":"core/#omniread.core.Content","title":"Content dataclass","text":"
    Content(\n    raw: bytes,\n    source: str,\n    content_type: ContentType | None = ...,\n    metadata: Mapping[str, Any] | None = ...,\n)\n

    Normalized representation of extracted content.

    Notes

    Responsibilities:

    - A `Content` instance represents a raw content payload along with\n  minimal contextual metadata describing its origin and type.\n- This class is the primary exchange format between scrapers,\n  parsers, and downstream consumers.\n
    "},{"location":"core/#omniread.core.Content-attributes","title":"Attributes","text":""},{"location":"core/#omniread.core.Content.content_type","title":"content_type class-attribute instance-attribute","text":"
    content_type: ContentType | None = None\n

    Optional MIME type of the content, if known.

    "},{"location":"core/#omniread.core.Content.metadata","title":"metadata class-attribute instance-attribute","text":"
    metadata: Mapping[str, Any] | None = None\n

    Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes).

    "},{"location":"core/#omniread.core.Content.raw","title":"raw instance-attribute","text":"
    raw: bytes\n

    Raw content bytes as retrieved from the source.

    "},{"location":"core/#omniread.core.Content.source","title":"source instance-attribute","text":"
    source: str\n

    Identifier of the content origin (URL, file path, or logical name).

    "},{"location":"core/#omniread.core.ContentType","title":"ContentType","text":"

    Bases: str, Enum

    Supported MIME types for extracted content.

    Notes

    Guarantees:

    - This enum represents the declared or inferred media type of the\n  content source.\n- It is primarily used for routing content to the appropriate\n  parser or downstream consumer.\n
    "},{"location":"core/#omniread.core.ContentType-attributes","title":"Attributes","text":""},{"location":"core/#omniread.core.ContentType.CSV","title":"CSV class-attribute instance-attribute","text":"
    CSV = 'text/csv'\n

    Comma-separated-value document content.

    "},{"location":"core/#omniread.core.ContentType.HTML","title":"HTML class-attribute instance-attribute","text":"
    HTML = 'text/html'\n

    HTML document content.

    "},{"location":"core/#omniread.core.ContentType.JSON","title":"JSON class-attribute instance-attribute","text":"
    JSON = 'application/json'\n

    JSON document content.

    "},{"location":"core/#omniread.core.ContentType.PDF","title":"PDF class-attribute instance-attribute","text":"
    PDF = 'application/pdf'\n

    PDF document content.

    "},{"location":"core/#omniread.core.ContentType.XLSX","title":"XLSX class-attribute instance-attribute","text":"
    XLSX = \"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet\"\n

    Office Open XML spreadsheet (xlsx/xlsm) content.

    "},{"location":"core/#omniread.core.ContentType.XML","title":"XML class-attribute instance-attribute","text":"
    XML = 'application/xml'\n

    XML document content.

    "},{"location":"core/content/","title":"Content","text":""},{"location":"core/content/#omniread.core.content","title":"omniread.core.content","text":""},{"location":"core/content/#omniread.core.content--summary","title":"Summary","text":"

    Canonical content models for OmniRead.

    This module defines the format-agnostic content representation used across all parsers and scrapers in OmniRead.

    The models defined here represent what was extracted, not how it was retrieved or parsed. Format-specific behavior and metadata must not alter the semantic meaning of these models.

    "},{"location":"core/content/#omniread.core.content-classes","title":"Classes","text":""},{"location":"core/content/#omniread.core.content.Content","title":"Content dataclass","text":"
    Content(\n    raw: bytes,\n    source: str,\n    content_type: ContentType | None = ...,\n    metadata: Mapping[str, Any] | None = ...,\n)\n

    Normalized representation of extracted content.

    Notes

    Responsibilities:

    - A `Content` instance represents a raw content payload along with\n  minimal contextual metadata describing its origin and type.\n- This class is the primary exchange format between scrapers,\n  parsers, and downstream consumers.\n
    "},{"location":"core/content/#omniread.core.content.Content-attributes","title":"Attributes","text":""},{"location":"core/content/#omniread.core.content.Content.content_type","title":"content_type class-attribute instance-attribute","text":"
    content_type: ContentType | None = None\n

    Optional MIME type of the content, if known.

    "},{"location":"core/content/#omniread.core.content.Content.metadata","title":"metadata class-attribute instance-attribute","text":"
    metadata: Mapping[str, Any] | None = None\n

    Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes).

    "},{"location":"core/content/#omniread.core.content.Content.raw","title":"raw instance-attribute","text":"
    raw: bytes\n

    Raw content bytes as retrieved from the source.

    "},{"location":"core/content/#omniread.core.content.Content.source","title":"source instance-attribute","text":"
    source: str\n

    Identifier of the content origin (URL, file path, or logical name).

    "},{"location":"core/content/#omniread.core.content.ContentType","title":"ContentType","text":"

    Bases: str, Enum

    Supported MIME types for extracted content.

    Notes

    Guarantees:

    - This enum represents the declared or inferred media type of the\n  content source.\n- It is primarily used for routing content to the appropriate\n  parser or downstream consumer.\n
    "},{"location":"core/content/#omniread.core.content.ContentType-attributes","title":"Attributes","text":""},{"location":"core/content/#omniread.core.content.ContentType.CSV","title":"CSV class-attribute instance-attribute","text":"
    CSV = 'text/csv'\n

    Comma-separated-value document content.

    "},{"location":"core/content/#omniread.core.content.ContentType.HTML","title":"HTML class-attribute instance-attribute","text":"
    HTML = 'text/html'\n

    HTML document content.

    "},{"location":"core/content/#omniread.core.content.ContentType.JSON","title":"JSON class-attribute instance-attribute","text":"
    JSON = 'application/json'\n

    JSON document content.

    "},{"location":"core/content/#omniread.core.content.ContentType.PDF","title":"PDF class-attribute instance-attribute","text":"
    PDF = 'application/pdf'\n

    PDF document content.

    "},{"location":"core/content/#omniread.core.content.ContentType.XLSX","title":"XLSX class-attribute instance-attribute","text":"
    XLSX = \"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet\"\n

    Office Open XML spreadsheet (xlsx/xlsm) content.

    "},{"location":"core/content/#omniread.core.content.ContentType.XML","title":"XML class-attribute instance-attribute","text":"
    XML = 'application/xml'\n

    XML document content.

    "},{"location":"core/parser/","title":"Parser","text":""},{"location":"core/parser/#omniread.core.parser","title":"omniread.core.parser","text":""},{"location":"core/parser/#omniread.core.parser--summary","title":"Summary","text":"

    Abstract parsing contracts for OmniRead.

    This module defines the format-agnostic parser interface used to transform raw content into structured, typed representations.

    Parsers are responsible for:

    • Interpreting a single Content instance
    • Validating compatibility with the content type
    • Producing a structured output suitable for downstream consumers

    Parsers are not responsible for:

    • Fetching or acquiring content
    • Performing retries or error recovery
    • Managing multiple content sources
    "},{"location":"core/parser/#omniread.core.parser-classes","title":"Classes","text":""},{"location":"core/parser/#omniread.core.parser.BaseParser","title":"BaseParser","text":"
    BaseParser(content: Content)\n

    Bases: ABC, Generic[T]

    Base interface for all parsers.

    Notes

    Guarantees:

    - A parser is a self-contained object that owns the `Content` it is\n  responsible for interpreting.\n- Consumers may rely on early validation of content compatibility\n  and type-stable return values from `parse()`.\n

    Responsibilities:

    - Implementations must declare supported content types via `supported_types`.\n- Implementations must raise parsing-specific exceptions from `parse()`.\n- Implementations must remain deterministic for a given input.\n

    Initialize the parser with content to be parsed.

    Parameters:

    Name Type Description Default content Content

    Content instance to be parsed.

    required

    Raises:

    Type Description ValueError

    If the content type is not supported by this parser.

    "},{"location":"core/parser/#omniread.core.parser.BaseParser-attributes","title":"Attributes","text":""},{"location":"core/parser/#omniread.core.parser.BaseParser.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = set()\n

    Set of content types supported by this parser. An empty set indicates that the parser is content-type agnostic.

    "},{"location":"core/parser/#omniread.core.parser.BaseParser-functions","title":"Functions","text":""},{"location":"core/parser/#omniread.core.parser.BaseParser.parse","title":"parse abstractmethod","text":"
    parse() -> T\n

    Parse the owned content into structured output.

    Returns:

    Name Type Description T T

    Parsed, structured representation.

    Raises:

    Type Description Exception

    Parsing-specific errors as defined by the implementation.

    Notes

    Responsibilities:

    - Implementations must fully consume the provided content and\n  return a deterministic, structured output.\n
    "},{"location":"core/parser/#omniread.core.parser.BaseParser.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"core/scraper/","title":"Scraper","text":""},{"location":"core/scraper/#omniread.core.scraper","title":"omniread.core.scraper","text":""},{"location":"core/scraper/#omniread.core.scraper--summary","title":"Summary","text":"

    Abstract scraping contracts for OmniRead.

    This module defines the format-agnostic scraper interface responsible for acquiring raw content from external sources.

    Scrapers are responsible for:

    • Locating and retrieving raw content bytes
    • Attaching minimal contextual metadata
    • Returning normalized Content objects

    Scrapers are explicitly NOT responsible for:

    • Parsing or interpreting content
    • Inferring structure or semantics
    • Performing content-type specific processing

    All interpretation must be delegated to parsers.

    "},{"location":"core/scraper/#omniread.core.scraper-classes","title":"Classes","text":""},{"location":"core/scraper/#omniread.core.scraper.BaseScraper","title":"BaseScraper","text":"

    Bases: ABC

    Base interface for all scrapers.

    Notes

    Responsibilities:

    - A scraper is responsible ONLY for fetching raw content (bytes)\n  from a source. It must not interpret or parse it.\n- A scraper is a stateless acquisition component that retrieves raw\n  content from a source and returns it as a `Content` object.\n- Scrapers define how content is obtained, not what the content means.\n- Implementations may vary in transport mechanism, authentication\n  strategy, retry and backoff behavior.\n

    Constraints:

    - Implementations must not parse content, modify content semantics,\n  or couple scraping logic to a specific parser.\n
    "},{"location":"core/scraper/#omniread.core.scraper.BaseScraper-functions","title":"Functions","text":""},{"location":"core/scraper/#omniread.core.scraper.BaseScraper.fetch","title":"fetch abstractmethod","text":"
    fetch(\n    source: str,\n    *,\n    metadata: Mapping[str, Any] | None = None\n) -> Content\n

    Fetch raw content from the given source.

    Parameters:

    Name Type Description Default source str

    Location identifier (URL, file path, S3 URI, etc.).

    required metadata Mapping[str, Any] | None

    Optional hints for the scraper (headers, auth, etc.).

    None

    Returns:

    Name Type Description Content Content

    Content object containing raw bytes and metadata.

    Raises:

    Type Description Exception

    Retrieval-specific errors as defined by the implementation.

    Notes

    Responsibilities:

    - Implementations must retrieve the content referenced by `source`\n  and return it as raw bytes wrapped in a `Content` object.\n
    "},{"location":"csv/","title":"Csv","text":""},{"location":"csv/#omniread.csv","title":"omniread.csv","text":""},{"location":"csv/#omniread.csv--summary","title":"Summary","text":"

    CSV subpackage for OmniRead.

    Provides acquisition and parsing of comma-separated-value content:

    • BaseCsvClient: abstract backing-store client for csv bytes.
    • FileSystemCsvClient: local filesystem implementation.
    • CsvScraper: wraps fetched bytes into canonical Content.
    • CsvParserBase: content-type-enforcing parser contract.
    • CsvParser: generic string-row parser built on the standard csv module.
    "},{"location":"csv/#omniread.csv-classes","title":"Classes","text":""},{"location":"csv/#omniread.csv.BaseCsvClient","title":"BaseCsvClient","text":"

    Bases: ABC

    Abstract client responsible for retrieving csv bytes.

    Retrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).

    Notes

    Responsibilities:

    - Implementations must accept a source identifier appropriate to\n  the backing store.\n- Return the full csv binary payload.\n- Raise retrieval-specific errors on failure.\n
    "},{"location":"csv/#omniread.csv.BaseCsvClient-functions","title":"Functions","text":""},{"location":"csv/#omniread.csv.BaseCsvClient.fetch","title":"fetch abstractmethod","text":"
    fetch(source: Any) -> bytes\n

    Fetch raw csv bytes from the given source.

    Parameters:

    Name Type Description Default source Any

    Identifier of the csv location, such as a file path, object storage key, or remote reference.

    required

    Returns:

    Name Type Description bytes bytes

    Raw csv bytes.

    Raises:

    Type Description Exception

    Retrieval-specific errors defined by the implementation.

    "},{"location":"csv/#omniread.csv.CsvParser","title":"CsvParser","text":"
    CsvParser(content: Content)\n

    Bases: CsvParserBase[list[list[str]]]

    Generic csv parser producing string rows from the document.

    Notes

    Responsibilities:

    - Decode the payload (UTF-8 with BOM support, Latin-1 fallback).\n- Detect the delimiter from a leading sample (`,` `;` tab `|`),\n  defaulting to `,`.\n- Normalize cells into deterministic stripped string values.\n- Expose row extraction helpers mirroring `XlsxParser.rows`.\n

    Constraints:

    - All values are strings; consumers requiring typed values must\n  convert on their side.\n- Quoted fields containing delimiters/newlines are handled by\n  the standard ``csv`` module.\n

    Initialize the parser.

    Parameters:

    Name Type Description Default content Content

    CSV content to parse; its type must be supported.

    required"},{"location":"csv/#omniread.csv.CsvParser-attributes","title":"Attributes","text":""},{"location":"csv/#omniread.csv.CsvParser.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {CSV}\n

    Set of content types supported by this parser (CSV only).

    "},{"location":"csv/#omniread.csv.CsvParser-functions","title":"Functions","text":""},{"location":"csv/#omniread.csv.CsvParser.parse","title":"parse","text":"
    parse() -> list[list[str]]\n

    Parse the document into normalized string rows.

    Returns:

    Type Description list[list[str]]

    list[list[str]]: Rows of the document.

    "},{"location":"csv/#omniread.csv.CsvParser.rows","title":"rows","text":"
    rows(*, skip_empty: bool = True) -> list[list[str]]\n

    Extract normalized string rows from the document.

    Parameters:

    Name Type Description Default skip_empty bool

    When True (default), rows whose cells are all blank are omitted.

    True

    Returns:

    Type Description list[list[str]]

    list[list[str]]: Normalized rows; trailing blank cells are trimmed per row.

    "},{"location":"csv/#omniread.csv.CsvParser.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"csv/#omniread.csv.CsvParserBase","title":"CsvParserBase","text":"
    CsvParserBase(content: Content)\n

    Bases: BaseParser[T], Generic[T]

    Base csv parser.

    Notes

    Responsibilities:

    - This class enforces csv content-type compatibility and provides\n  the extension point for implementing concrete csv parsing\n  strategies.\n

    Constraints:

    - Concrete implementations must define the output type `T` and\n  implement the `parse()` method.\n

    Initialize the parser with content to be parsed.

    Parameters:

    Name Type Description Default content Content

    Content instance to be parsed.

    required

    Raises:

    Type Description ValueError

    If the content type is not supported by this parser.

    "},{"location":"csv/#omniread.csv.CsvParserBase-attributes","title":"Attributes","text":""},{"location":"csv/#omniread.csv.CsvParserBase.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {CSV}\n

    Set of content types supported by this parser (CSV only).

    "},{"location":"csv/#omniread.csv.CsvParserBase-functions","title":"Functions","text":""},{"location":"csv/#omniread.csv.CsvParserBase.parse","title":"parse abstractmethod","text":"
    parse() -> T\n

    Parse csv content into a structured output.

    Returns:

    Name Type Description T T

    Parsed representation of type T.

    Raises:

    Type Description Exception

    Parsing-specific errors as defined by the implementation.

    "},{"location":"csv/#omniread.csv.CsvParserBase.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"csv/#omniread.csv.CsvScraper","title":"CsvScraper","text":"
    CsvScraper(*, client: BaseCsvClient)\n

    Scraper for csv documents.

    Notes

    Responsibilities:

    - Fetch raw csv bytes via the configured client.\n- Wrap the payload in a canonical `Content` instance with the\n  CSV content type and source identifier.\n

    Constraints:

    - The scraper does not perform parsing or interpretation.\n- Does not assume a specific storage backend.\n

    Initialize the CSV scraper.

    Parameters:

    Name Type Description Default client BaseCsvClient

    Client responsible for retrieving raw csv bytes.

    required"},{"location":"csv/#omniread.csv.CsvScraper-functions","title":"Functions","text":""},{"location":"csv/#omniread.csv.CsvScraper.fetch","title":"fetch","text":"
    fetch(\n    source: Any,\n    *,\n    metadata: Mapping[str, Any] | None = None\n) -> Content\n

    Fetch a csv document from the given source.

    Parameters:

    Name Type Description Default source Any

    Identifier of the csv source as understood by the configured client.

    required metadata Mapping[str, Any] | None

    Optional metadata to attach to the returned content.

    None

    Returns:

    Name Type Description Content Content

    A Content instance containing raw csv bytes, source identifier, CSV content type, and optional metadata.

    Raises:

    Type Description Exception

    Retrieval-specific errors raised by the client.

    "},{"location":"csv/#omniread.csv.FileSystemCsvClient","title":"FileSystemCsvClient","text":"

    Bases: BaseCsvClient

    CSV client that reads from the local filesystem.

    Notes

    Guarantees:

    - This client reads csv files directly from the disk and\n  returns their raw binary contents.\n
    "},{"location":"csv/#omniread.csv.FileSystemCsvClient-functions","title":"Functions","text":""},{"location":"csv/#omniread.csv.FileSystemCsvClient.fetch","title":"fetch","text":"
    fetch(path: Path) -> bytes\n

    Read a csv file from the local filesystem.

    Parameters:

    Name Type Description Default path Path

    Filesystem path to the csv file.

    required

    Returns:

    Name Type Description bytes bytes

    Raw csv bytes.

    Raises:

    Type Description FileNotFoundError

    If the path does not exist.

    ValueError

    If the path exists but is not a file.

    "},{"location":"csv/client/","title":"Client","text":""},{"location":"csv/client/#omniread.csv.client","title":"omniread.csv.client","text":""},{"location":"csv/client/#omniread.csv.client--summary","title":"Summary","text":"

    CSV client abstractions for OmniRead.

    This module defines the client layer responsible for retrieving raw comma-separated-value document bytes from a concrete backing store.

    Clients provide low-level access to csv binaries and are intentionally decoupled from scraping and parsing logic. They do not perform validation, interpretation, or content extraction.

    Typical backing stores include:

    • Local filesystems
    • Object storage (S3, GCS, etc.)
    • Network file systems
    "},{"location":"csv/client/#omniread.csv.client-classes","title":"Classes","text":""},{"location":"csv/client/#omniread.csv.client.BaseCsvClient","title":"BaseCsvClient","text":"

    Bases: ABC

    Abstract client responsible for retrieving csv bytes.

    Retrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).

    Notes

    Responsibilities:

    - Implementations must accept a source identifier appropriate to\n  the backing store.\n- Return the full csv binary payload.\n- Raise retrieval-specific errors on failure.\n
    "},{"location":"csv/client/#omniread.csv.client.BaseCsvClient-functions","title":"Functions","text":""},{"location":"csv/client/#omniread.csv.client.BaseCsvClient.fetch","title":"fetch abstractmethod","text":"
    fetch(source: Any) -> bytes\n

    Fetch raw csv bytes from the given source.

    Parameters:

    Name Type Description Default source Any

    Identifier of the csv location, such as a file path, object storage key, or remote reference.

    required

    Returns:

    Name Type Description bytes bytes

    Raw csv bytes.

    Raises:

    Type Description Exception

    Retrieval-specific errors defined by the implementation.

    "},{"location":"csv/client/#omniread.csv.client.FileSystemCsvClient","title":"FileSystemCsvClient","text":"

    Bases: BaseCsvClient

    CSV client that reads from the local filesystem.

    Notes

    Guarantees:

    - This client reads csv files directly from the disk and\n  returns their raw binary contents.\n
    "},{"location":"csv/client/#omniread.csv.client.FileSystemCsvClient-functions","title":"Functions","text":""},{"location":"csv/client/#omniread.csv.client.FileSystemCsvClient.fetch","title":"fetch","text":"
    fetch(path: Path) -> bytes\n

    Read a csv file from the local filesystem.

    Parameters:

    Name Type Description Default path Path

    Filesystem path to the csv file.

    required

    Returns:

    Name Type Description bytes bytes

    Raw csv bytes.

    Raises:

    Type Description FileNotFoundError

    If the path does not exist.

    ValueError

    If the path exists but is not a file.

    "},{"location":"csv/parser/","title":"Parser","text":""},{"location":"csv/parser/#omniread.csv.parser","title":"omniread.csv.parser","text":""},{"location":"csv/parser/#omniread.csv.parser--summary","title":"Summary","text":"

    CSV parser implementations for OmniRead.

    This module provides a concrete, generic parser for comma-separated-value documents. It exposes records as lists of string cells so downstream consumers can interpret tabular content without depending on the csv module directly.

    The parser is intentionally statement-agnostic: it performs no header detection or column interpretation beyond basic cell normalization and delimiter detection.

    "},{"location":"csv/parser/#omniread.csv.parser-classes","title":"Classes","text":""},{"location":"csv/parser/#omniread.csv.parser.CsvParser","title":"CsvParser","text":"
    CsvParser(content: Content)\n

    Bases: CsvParserBase[list[list[str]]]

    Generic csv parser producing string rows from the document.

    Notes

    Responsibilities:

    - Decode the payload (UTF-8 with BOM support, Latin-1 fallback).\n- Detect the delimiter from a leading sample (`,` `;` tab `|`),\n  defaulting to `,`.\n- Normalize cells into deterministic stripped string values.\n- Expose row extraction helpers mirroring `XlsxParser.rows`.\n

    Constraints:

    - All values are strings; consumers requiring typed values must\n  convert on their side.\n- Quoted fields containing delimiters/newlines are handled by\n  the standard ``csv`` module.\n

    Initialize the parser.

    Parameters:

    Name Type Description Default content Content

    CSV content to parse; its type must be supported.

    required"},{"location":"csv/parser/#omniread.csv.parser.CsvParser-attributes","title":"Attributes","text":""},{"location":"csv/parser/#omniread.csv.parser.CsvParser.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {CSV}\n

    Set of content types supported by this parser (CSV only).

    "},{"location":"csv/parser/#omniread.csv.parser.CsvParser-functions","title":"Functions","text":""},{"location":"csv/parser/#omniread.csv.parser.CsvParser.parse","title":"parse","text":"
    parse() -> list[list[str]]\n

    Parse the document into normalized string rows.

    Returns:

    Type Description list[list[str]]

    list[list[str]]: Rows of the document.

    "},{"location":"csv/parser/#omniread.csv.parser.CsvParser.rows","title":"rows","text":"
    rows(*, skip_empty: bool = True) -> list[list[str]]\n

    Extract normalized string rows from the document.

    Parameters:

    Name Type Description Default skip_empty bool

    When True (default), rows whose cells are all blank are omitted.

    True

    Returns:

    Type Description list[list[str]]

    list[list[str]]: Normalized rows; trailing blank cells are trimmed per row.

    "},{"location":"csv/parser/#omniread.csv.parser.CsvParser.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"csv/parser_base/","title":"Parser Base","text":""},{"location":"csv/parser_base/#omniread.csv.parser_base","title":"omniread.csv.parser_base","text":""},{"location":"csv/parser_base/#omniread.csv.parser_base--summary","title":"Summary","text":"

    CSV parser base implementation for OmniRead.

    This module defines the CSV-specific parser contract, extending the format-agnostic BaseParser with constraints appropriate for comma-separated-value documents.

    "},{"location":"csv/parser_base/#omniread.csv.parser_base-classes","title":"Classes","text":""},{"location":"csv/parser_base/#omniread.csv.parser_base.CsvParserBase","title":"CsvParserBase","text":"
    CsvParserBase(content: Content)\n

    Bases: BaseParser[T], Generic[T]

    Base csv parser.

    Notes

    Responsibilities:

    - This class enforces csv content-type compatibility and provides\n  the extension point for implementing concrete csv parsing\n  strategies.\n

    Constraints:

    - Concrete implementations must define the output type `T` and\n  implement the `parse()` method.\n

    Initialize the parser with content to be parsed.

    Parameters:

    Name Type Description Default content Content

    Content instance to be parsed.

    required

    Raises:

    Type Description ValueError

    If the content type is not supported by this parser.

    "},{"location":"csv/parser_base/#omniread.csv.parser_base.CsvParserBase-attributes","title":"Attributes","text":""},{"location":"csv/parser_base/#omniread.csv.parser_base.CsvParserBase.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {CSV}\n

    Set of content types supported by this parser (CSV only).

    "},{"location":"csv/parser_base/#omniread.csv.parser_base.CsvParserBase-functions","title":"Functions","text":""},{"location":"csv/parser_base/#omniread.csv.parser_base.CsvParserBase.parse","title":"parse abstractmethod","text":"
    parse() -> T\n

    Parse csv content into a structured output.

    Returns:

    Name Type Description T T

    Parsed representation of type T.

    Raises:

    Type Description Exception

    Parsing-specific errors as defined by the implementation.

    "},{"location":"csv/parser_base/#omniread.csv.parser_base.CsvParserBase.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"csv/scraper/","title":"Scraper","text":""},{"location":"csv/scraper/#omniread.csv.scraper","title":"omniread.csv.scraper","text":""},{"location":"csv/scraper/#omniread.csv.scraper--summary","title":"Summary","text":"

    CSV scraper for OmniRead.

    This module defines the scraper responsible for acquiring raw comma-separated-value document content from a backing store via a configured client.

    The scraper does not interpret or parse the acquired bytes; it wraps them in the canonical Content model.

    "},{"location":"csv/scraper/#omniread.csv.scraper-classes","title":"Classes","text":""},{"location":"csv/scraper/#omniread.csv.scraper.CsvScraper","title":"CsvScraper","text":"
    CsvScraper(*, client: BaseCsvClient)\n

    Scraper for csv documents.

    Notes

    Responsibilities:

    - Fetch raw csv bytes via the configured client.\n- Wrap the payload in a canonical `Content` instance with the\n  CSV content type and source identifier.\n

    Constraints:

    - The scraper does not perform parsing or interpretation.\n- Does not assume a specific storage backend.\n

    Initialize the CSV scraper.

    Parameters:

    Name Type Description Default client BaseCsvClient

    Client responsible for retrieving raw csv bytes.

    required"},{"location":"csv/scraper/#omniread.csv.scraper.CsvScraper-functions","title":"Functions","text":""},{"location":"csv/scraper/#omniread.csv.scraper.CsvScraper.fetch","title":"fetch","text":"
    fetch(\n    source: Any,\n    *,\n    metadata: Mapping[str, Any] | None = None\n) -> Content\n

    Fetch a csv document from the given source.

    Parameters:

    Name Type Description Default source Any

    Identifier of the csv source as understood by the configured client.

    required metadata Mapping[str, Any] | None

    Optional metadata to attach to the returned content.

    None

    Returns:

    Name Type Description Content Content

    A Content instance containing raw csv bytes, source identifier, CSV content type, and optional metadata.

    Raises:

    Type Description Exception

    Retrieval-specific errors raised by the client.

    "},{"location":"html/","title":"Html","text":""},{"location":"html/#omniread.html","title":"omniread.html","text":""},{"location":"html/#omniread.html--summary","title":"Summary","text":"

    HTML format implementation for OmniRead.

    This package provides HTML-specific implementations of the core OmniRead contracts defined in omniread.core.

    It includes:

    • HTML parsers that interpret HTML content.
    • HTML scrapers that retrieve HTML documents.

    Key characteristics:

    • Implements, but does not redefine, core contracts.
    • May contain HTML-specific behavior and edge-case handling.
    • Produces canonical content models defined in omniread.core.content.

    Consumers should depend on omniread.core interfaces wherever possible and use this package only when HTML-specific behavior is required.

    "},{"location":"html/#omniread.html--public-api","title":"Public API","text":"
    • HTMLScraper
    • HTMLParser
    "},{"location":"html/#omniread.html-classes","title":"Classes","text":""},{"location":"html/#omniread.html.HTMLParser","title":"HTMLParser","text":"
    HTMLParser(content: Content, features: str = 'html.parser')\n

    Bases: BaseParser[T], Generic[T]

    Base HTML parser.

    Notes

    Responsibilities:

    - This class extends the core `BaseParser` with HTML-specific behavior,\n  including DOM parsing via BeautifulSoup and reusable extraction helpers.\n- Provides reusable helpers for HTML extraction. Concrete parsers must\n  explicitly define the return type.\n

    Guarantees:

    - Accepts only HTML content.\n- Owns a parsed BeautifulSoup DOM tree.\n- Provides pure helper utilities for common HTML structures.\n

    Constraints:

    - Concrete subclasses must define the output type `T` and implement\n  the `parse()` method.\n

    Initialize the HTML parser.

    Parameters:

    Name Type Description Default content Content

    HTML content to be parsed.

    required features str

    BeautifulSoup parser backend to use (e.g., 'html.parser', 'lxml').

    'html.parser'

    Raises:

    Type Description ValueError

    If the content is empty or not valid HTML.

    "},{"location":"html/#omniread.html.HTMLParser-attributes","title":"Attributes","text":""},{"location":"html/#omniread.html.HTMLParser.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {HTML}\n

    Set of content types supported by this parser (HTML only).

    "},{"location":"html/#omniread.html.HTMLParser-functions","title":"Functions","text":""},{"location":"html/#omniread.html.HTMLParser.parse","title":"parse abstractmethod","text":"
    parse() -> T\n

    Fully parse the HTML content into structured output.

    Returns:

    Name Type Description T T

    Parsed representation of type T.

    Notes

    Responsibilities:

    - Implementations must fully interpret the HTML DOM and return a\n  deterministic, structured output.\n
    "},{"location":"html/#omniread.html.HTMLParser.parse_div","title":"parse_div staticmethod","text":"
    parse_div(div: Tag, *, separator: str = ' ') -> str\n

    Extract normalized text from a <div> element.

    Parameters:

    Name Type Description Default div Tag

    BeautifulSoup tag representing a <div>.

    required separator str

    String used to separate text nodes.

    ' '

    Returns:

    Name Type Description str str

    Flattened, whitespace-normalized text content.

    "},{"location":"html/#omniread.html.HTMLParser.parse_link","title":"parse_link staticmethod","text":"
    parse_link(a: Tag) -> str | None\n

    Extract the hyperlink reference from an <a> element.

    Parameters:

    Name Type Description Default a Tag

    BeautifulSoup tag representing an anchor.

    required

    Returns:

    Type Description str | None

    str | None: The value of the href attribute, or None if absent.

    "},{"location":"html/#omniread.html.HTMLParser.parse_meta","title":"parse_meta","text":"
    parse_meta() -> dict[str, Any]\n

    Extract high-level metadata from the HTML document.

    Returns:

    Type Description dict[str, Any]

    dict[str, Any]: Dictionary containing extracted metadata.

    Notes

    Responsibilities:

    - Extract high-level metadata from the HTML document.\n- This includes: Document title, `<meta>` tag name/property to\n  content mappings.\n
    "},{"location":"html/#omniread.html.HTMLParser.parse_table","title":"parse_table staticmethod","text":"
    parse_table(table: Tag) -> list[list[str]]\n

    Parse an HTML table into a 2D list of strings.

    Parameters:

    Name Type Description Default table Tag

    BeautifulSoup tag representing a <table>.

    required

    Returns:

    Type Description list[list[str]]

    list[list[str]]: A list of rows, where each row is a list of cell text values.

    "},{"location":"html/#omniread.html.HTMLParser.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"html/#omniread.html.HTMLScraper","title":"HTMLScraper","text":"
    HTMLScraper(\n    *,\n    client: httpx.Client | None = None,\n    timeout: float = 15.0,\n    headers: Mapping[str, str] | None = None,\n    follow_redirects: bool = True\n)\n

    Bases: BaseScraper

    Base HTML scraper using httpx.

    Notes

    Responsibilities:

    - This scraper retrieves HTML documents over HTTP(S) and returns\n  them as raw content wrapped in a `Content` object.\n- Fetches raw bytes and metadata only.\n- The scraper uses `httpx.Client` for HTTP requests, enforces an\n  HTML content type, and preserves HTTP response metadata.\n

    Constraints:

    - The scraper does not: Parse HTML, perform retries or backoff,\n  handle non-HTML responses.\n

    Initialize the HTML scraper.

    Parameters:

    Name Type Description Default client Client | None

    Optional pre-configured httpx.Client. If omitted, a client is created internally.

    None timeout float

    Request timeout in seconds.

    15.0 headers Mapping[str, str] | None

    Optional default HTTP headers.

    None follow_redirects bool

    Whether to follow HTTP redirects.

    True"},{"location":"html/#omniread.html.HTMLScraper-functions","title":"Functions","text":""},{"location":"html/#omniread.html.HTMLScraper.fetch","title":"fetch","text":"
    fetch(\n    source: str,\n    *,\n    metadata: Mapping[str, Any] | None = None\n) -> Content\n

    Fetch an HTML document from the given source.

    Parameters:

    Name Type Description Default source str

    URL of the HTML document.

    required metadata Mapping[str, Any] | None

    Optional metadata to be merged into the returned content.

    None

    Returns:

    Name Type Description Content Content

    A Content instance containing raw HTML bytes, source URL, HTML content type, and HTTP response metadata.

    Raises:

    Type Description HTTPError

    If the HTTP request fails.

    ValueError

    If the response is not valid HTML.

    "},{"location":"html/#omniread.html.HTMLScraper.validate_content_type","title":"validate_content_type","text":"
    validate_content_type(response: httpx.Response) -> None\n

    Validate that the HTTP response contains HTML content.

    Parameters:

    Name Type Description Default response Response

    HTTP response returned by httpx.

    required

    Raises:

    Type Description ValueError

    If the Content-Type header is missing or does not indicate HTML content.

    "},{"location":"html/parser/","title":"Parser","text":""},{"location":"html/parser/#omniread.html.parser","title":"omniread.html.parser","text":""},{"location":"html/parser/#omniread.html.parser--summary","title":"Summary","text":"

    HTML parser base implementations for OmniRead.

    This module provides reusable HTML parsing utilities built on top of the abstract parser contracts defined in omniread.core.parser.

    It supplies:

    • Content-type enforcement for HTML inputs
    • BeautifulSoup initialization and lifecycle management
    • Common helper methods for extracting structured data from HTML elements

    Concrete parsers must subclass HTMLParser and implement the parse() method to return a structured representation appropriate for their use case.

    "},{"location":"html/parser/#omniread.html.parser-classes","title":"Classes","text":""},{"location":"html/parser/#omniread.html.parser.HTMLParser","title":"HTMLParser","text":"
    HTMLParser(content: Content, features: str = 'html.parser')\n

    Bases: BaseParser[T], Generic[T]

    Base HTML parser.

    Notes

    Responsibilities:

    - This class extends the core `BaseParser` with HTML-specific behavior,\n  including DOM parsing via BeautifulSoup and reusable extraction helpers.\n- Provides reusable helpers for HTML extraction. Concrete parsers must\n  explicitly define the return type.\n

    Guarantees:

    - Accepts only HTML content.\n- Owns a parsed BeautifulSoup DOM tree.\n- Provides pure helper utilities for common HTML structures.\n

    Constraints:

    - Concrete subclasses must define the output type `T` and implement\n  the `parse()` method.\n

    Initialize the HTML parser.

    Parameters:

    Name Type Description Default content Content

    HTML content to be parsed.

    required features str

    BeautifulSoup parser backend to use (e.g., 'html.parser', 'lxml').

    'html.parser'

    Raises:

    Type Description ValueError

    If the content is empty or not valid HTML.

    "},{"location":"html/parser/#omniread.html.parser.HTMLParser-attributes","title":"Attributes","text":""},{"location":"html/parser/#omniread.html.parser.HTMLParser.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {HTML}\n

    Set of content types supported by this parser (HTML only).

    "},{"location":"html/parser/#omniread.html.parser.HTMLParser-functions","title":"Functions","text":""},{"location":"html/parser/#omniread.html.parser.HTMLParser.parse","title":"parse abstractmethod","text":"
    parse() -> T\n

    Fully parse the HTML content into structured output.

    Returns:

    Name Type Description T T

    Parsed representation of type T.

    Notes

    Responsibilities:

    - Implementations must fully interpret the HTML DOM and return a\n  deterministic, structured output.\n
    "},{"location":"html/parser/#omniread.html.parser.HTMLParser.parse_div","title":"parse_div staticmethod","text":"
    parse_div(div: Tag, *, separator: str = ' ') -> str\n

    Extract normalized text from a <div> element.

    Parameters:

    Name Type Description Default div Tag

    BeautifulSoup tag representing a <div>.

    required separator str

    String used to separate text nodes.

    ' '

    Returns:

    Name Type Description str str

    Flattened, whitespace-normalized text content.

    "},{"location":"html/parser/#omniread.html.parser.HTMLParser.parse_link","title":"parse_link staticmethod","text":"
    parse_link(a: Tag) -> str | None\n

    Extract the hyperlink reference from an <a> element.

    Parameters:

    Name Type Description Default a Tag

    BeautifulSoup tag representing an anchor.

    required

    Returns:

    Type Description str | None

    str | None: The value of the href attribute, or None if absent.

    "},{"location":"html/parser/#omniread.html.parser.HTMLParser.parse_meta","title":"parse_meta","text":"
    parse_meta() -> dict[str, Any]\n

    Extract high-level metadata from the HTML document.

    Returns:

    Type Description dict[str, Any]

    dict[str, Any]: Dictionary containing extracted metadata.

    Notes

    Responsibilities:

    - Extract high-level metadata from the HTML document.\n- This includes: Document title, `<meta>` tag name/property to\n  content mappings.\n
    "},{"location":"html/parser/#omniread.html.parser.HTMLParser.parse_table","title":"parse_table staticmethod","text":"
    parse_table(table: Tag) -> list[list[str]]\n

    Parse an HTML table into a 2D list of strings.

    Parameters:

    Name Type Description Default table Tag

    BeautifulSoup tag representing a <table>.

    required

    Returns:

    Type Description list[list[str]]

    list[list[str]]: A list of rows, where each row is a list of cell text values.

    "},{"location":"html/parser/#omniread.html.parser.HTMLParser.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"html/scraper/","title":"Scraper","text":""},{"location":"html/scraper/#omniread.html.scraper","title":"omniread.html.scraper","text":""},{"location":"html/scraper/#omniread.html.scraper--summary","title":"Summary","text":"

    HTML scraping implementation for OmniRead.

    This module provides an HTTP-based scraper for retrieving HTML documents. It implements the core BaseScraper contract using httpx as the transport layer.

    This scraper is responsible for:

    • Fetching raw HTML bytes over HTTP(S)
    • Validating response content type
    • Attaching HTTP metadata to the returned content

    This scraper is not responsible for:

    • Parsing or interpreting HTML
    • Retrying failed requests
    • Managing crawl policies or rate limiting
    "},{"location":"html/scraper/#omniread.html.scraper-classes","title":"Classes","text":""},{"location":"html/scraper/#omniread.html.scraper.HTMLScraper","title":"HTMLScraper","text":"
    HTMLScraper(\n    *,\n    client: httpx.Client | None = None,\n    timeout: float = 15.0,\n    headers: Mapping[str, str] | None = None,\n    follow_redirects: bool = True\n)\n

    Bases: BaseScraper

    Base HTML scraper using httpx.

    Notes

    Responsibilities:

    - This scraper retrieves HTML documents over HTTP(S) and returns\n  them as raw content wrapped in a `Content` object.\n- Fetches raw bytes and metadata only.\n- The scraper uses `httpx.Client` for HTTP requests, enforces an\n  HTML content type, and preserves HTTP response metadata.\n

    Constraints:

    - The scraper does not: Parse HTML, perform retries or backoff,\n  handle non-HTML responses.\n

    Initialize the HTML scraper.

    Parameters:

    Name Type Description Default client Client | None

    Optional pre-configured httpx.Client. If omitted, a client is created internally.

    None timeout float

    Request timeout in seconds.

    15.0 headers Mapping[str, str] | None

    Optional default HTTP headers.

    None follow_redirects bool

    Whether to follow HTTP redirects.

    True"},{"location":"html/scraper/#omniread.html.scraper.HTMLScraper-functions","title":"Functions","text":""},{"location":"html/scraper/#omniread.html.scraper.HTMLScraper.fetch","title":"fetch","text":"
    fetch(\n    source: str,\n    *,\n    metadata: Mapping[str, Any] | None = None\n) -> Content\n

    Fetch an HTML document from the given source.

    Parameters:

    Name Type Description Default source str

    URL of the HTML document.

    required metadata Mapping[str, Any] | None

    Optional metadata to be merged into the returned content.

    None

    Returns:

    Name Type Description Content Content

    A Content instance containing raw HTML bytes, source URL, HTML content type, and HTTP response metadata.

    Raises:

    Type Description HTTPError

    If the HTTP request fails.

    ValueError

    If the response is not valid HTML.

    "},{"location":"html/scraper/#omniread.html.scraper.HTMLScraper.validate_content_type","title":"validate_content_type","text":"
    validate_content_type(response: httpx.Response) -> None\n

    Validate that the HTTP response contains HTML content.

    Parameters:

    Name Type Description Default response Response

    HTTP response returned by httpx.

    required

    Raises:

    Type Description ValueError

    If the Content-Type header is missing or does not indicate HTML content.

    "},{"location":"omniread/","title":"Omniread","text":"
    • Core
    • Csv
    • Html
    • Pdf
    • Xlsx
    "},{"location":"omniread/#omniread","title":"omniread","text":""},{"location":"omniread/#omniread--summary","title":"Summary","text":"

    OmniRead \u2014 format-agnostic content acquisition and parsing framework.

    OmniRead provides a cleanly layered architecture for fetching, parsing, and normalizing content from heterogeneous sources such as HTML documents and PDF files.

    The library is structured around three core concepts:

    1. Content: A canonical, format-agnostic container representing raw content bytes and minimal contextual metadata.
    2. Scrapers: Components responsible for acquiring raw content from a source (HTTP, filesystem, object storage, etc.). Scrapers never interpret content.
    3. Parsers: Components responsible for interpreting acquired content and converting it into structured, typed representations.

    OmniRead deliberately separates these responsibilities to ensure:

    • Clear boundaries between IO and interpretation.
    • Replaceable implementations per format.
    • Predictable, testable behavior.
    "},{"location":"omniread/#omniread--installation","title":"Installation","text":"

    Install OmniRead using pip:

    pip install omniread\n

    Install OmniRead using Poetry:

    poetry add omniread\n

    "},{"location":"omniread/#omniread--quick-start","title":"Quick start","text":"Example

    HTML example:

    from omniread import HTMLScraper, HTMLParser\n\nscraper = HTMLScraper()\ncontent = scraper.fetch(\"https://example.com\")\n\nclass TitleParser(HTMLParser[str]):\n    def parse(self) -> str:\n        return self._soup.title.string\n\nparser = TitleParser(content)\ntitle = parser.parse()\n

    PDF example:

    from omniread import FileSystemPDFClient, PDFScraper, PDFParser\nfrom pathlib import Path\n\nclient = FileSystemPDFClient()\nscraper = PDFScraper(client=client)\ncontent = scraper.fetch(Path(\"document.pdf\"))\n\nclass TextPDFParser(PDFParser[str]):\n    def parse(self) -> str:\n        # implement PDF text extraction\n        ...\n\nparser = TextPDFParser(content)\nresult = parser.parse()\n

    "},{"location":"omniread/#omniread--public-api","title":"Public API","text":"

    This module re-exports the recommended public entry points of OmniRead. Consumers are encouraged to import from this namespace rather than from format-specific submodules directly, unless advanced customization is required.

    • Content: Canonical content model.
    • ContentType: Supported media types.
    • HTMLScraper: HTTP-based HTML acquisition.
    • HTMLParser: Base parser for HTML DOM interpretation.
    • FileSystemPDFClient: Local filesystem PDF access.
    • PDFScraper: PDF-specific content acquisition.
    • PDFParser: Base parser for PDF binary interpretation.
    • FileSystemXlsxClient: Local filesystem spreadsheet access.
    • XlsxScraper: XLSX-specific content acquisition.
    • XlsxParser: Generic string-row parser for xlsx workbooks.
    "},{"location":"omniread/#omniread--core-philosophy","title":"Core Philosophy","text":"

    OmniRead is designed as a decoupled content engine:

    1. Separation of Concerns: Scrapers fetch, Parsers interpret. Neither knows about the other.
    2. Normalized Exchange: All components communicate via the Content model, ensuring a consistent contract.
    3. Format Agnosticism: The core logic is independent of whether the input is HTML, PDF, or JSON.
    "},{"location":"omniread/#omniread-classes","title":"Classes","text":""},{"location":"omniread/#omniread.Content","title":"Content dataclass","text":"
    Content(\n    raw: bytes,\n    source: str,\n    content_type: ContentType | None = ...,\n    metadata: Mapping[str, Any] | None = ...,\n)\n

    Normalized representation of extracted content.

    Notes

    Responsibilities:

    - A `Content` instance represents a raw content payload along with\n  minimal contextual metadata describing its origin and type.\n- This class is the primary exchange format between scrapers,\n  parsers, and downstream consumers.\n
    "},{"location":"omniread/#omniread.Content-attributes","title":"Attributes","text":""},{"location":"omniread/#omniread.Content.content_type","title":"content_type class-attribute instance-attribute","text":"
    content_type: ContentType | None = None\n

    Optional MIME type of the content, if known.

    "},{"location":"omniread/#omniread.Content.metadata","title":"metadata class-attribute instance-attribute","text":"
    metadata: Mapping[str, Any] | None = None\n

    Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes).

    "},{"location":"omniread/#omniread.Content.raw","title":"raw instance-attribute","text":"
    raw: bytes\n

    Raw content bytes as retrieved from the source.

    "},{"location":"omniread/#omniread.Content.source","title":"source instance-attribute","text":"
    source: str\n

    Identifier of the content origin (URL, file path, or logical name).

    "},{"location":"omniread/#omniread.ContentType","title":"ContentType","text":"

    Bases: str, Enum

    Supported MIME types for extracted content.

    Notes

    Guarantees:

    - This enum represents the declared or inferred media type of the\n  content source.\n- It is primarily used for routing content to the appropriate\n  parser or downstream consumer.\n
    "},{"location":"omniread/#omniread.ContentType-attributes","title":"Attributes","text":""},{"location":"omniread/#omniread.ContentType.CSV","title":"CSV class-attribute instance-attribute","text":"
    CSV = 'text/csv'\n

    Comma-separated-value document content.

    "},{"location":"omniread/#omniread.ContentType.HTML","title":"HTML class-attribute instance-attribute","text":"
    HTML = 'text/html'\n

    HTML document content.

    "},{"location":"omniread/#omniread.ContentType.JSON","title":"JSON class-attribute instance-attribute","text":"
    JSON = 'application/json'\n

    JSON document content.

    "},{"location":"omniread/#omniread.ContentType.PDF","title":"PDF class-attribute instance-attribute","text":"
    PDF = 'application/pdf'\n

    PDF document content.

    "},{"location":"omniread/#omniread.ContentType.XLSX","title":"XLSX class-attribute instance-attribute","text":"
    XLSX = \"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet\"\n

    Office Open XML spreadsheet (xlsx/xlsm) content.

    "},{"location":"omniread/#omniread.ContentType.XML","title":"XML class-attribute instance-attribute","text":"
    XML = 'application/xml'\n

    XML document content.

    "},{"location":"omniread/#omniread.FileSystemPDFClient","title":"FileSystemPDFClient","text":"

    Bases: BasePDFClient

    PDF client that reads from the local filesystem.

    Notes

    Guarantees:

    - This client reads PDF files directly from the disk and returns\n  their raw binary contents.\n
    "},{"location":"omniread/#omniread.FileSystemPDFClient-functions","title":"Functions","text":""},{"location":"omniread/#omniread.FileSystemPDFClient.fetch","title":"fetch","text":"
    fetch(path: Path) -> bytes\n

    Read a PDF file from the local filesystem.

    Parameters:

    Name Type Description Default path Path

    Filesystem path to the PDF file.

    required

    Returns:

    Name Type Description bytes bytes

    Raw PDF bytes.

    Raises:

    Type Description FileNotFoundError

    If the path does not exist.

    ValueError

    If the path exists but is not a file.

    "},{"location":"omniread/#omniread.HTMLParser","title":"HTMLParser","text":"
    HTMLParser(content: Content, features: str = 'html.parser')\n

    Bases: BaseParser[T], Generic[T]

    Base HTML parser.

    Notes

    Responsibilities:

    - This class extends the core `BaseParser` with HTML-specific behavior,\n  including DOM parsing via BeautifulSoup and reusable extraction helpers.\n- Provides reusable helpers for HTML extraction. Concrete parsers must\n  explicitly define the return type.\n

    Guarantees:

    - Accepts only HTML content.\n- Owns a parsed BeautifulSoup DOM tree.\n- Provides pure helper utilities for common HTML structures.\n

    Constraints:

    - Concrete subclasses must define the output type `T` and implement\n  the `parse()` method.\n

    Initialize the HTML parser.

    Parameters:

    Name Type Description Default content Content

    HTML content to be parsed.

    required features str

    BeautifulSoup parser backend to use (e.g., 'html.parser', 'lxml').

    'html.parser'

    Raises:

    Type Description ValueError

    If the content is empty or not valid HTML.

    "},{"location":"omniread/#omniread.HTMLParser-attributes","title":"Attributes","text":""},{"location":"omniread/#omniread.HTMLParser.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {HTML}\n

    Set of content types supported by this parser (HTML only).

    "},{"location":"omniread/#omniread.HTMLParser-functions","title":"Functions","text":""},{"location":"omniread/#omniread.HTMLParser.parse","title":"parse abstractmethod","text":"
    parse() -> T\n

    Fully parse the HTML content into structured output.

    Returns:

    Name Type Description T T

    Parsed representation of type T.

    Notes

    Responsibilities:

    - Implementations must fully interpret the HTML DOM and return a\n  deterministic, structured output.\n
    "},{"location":"omniread/#omniread.HTMLParser.parse_div","title":"parse_div staticmethod","text":"
    parse_div(div: Tag, *, separator: str = ' ') -> str\n

    Extract normalized text from a <div> element.

    Parameters:

    Name Type Description Default div Tag

    BeautifulSoup tag representing a <div>.

    required separator str

    String used to separate text nodes.

    ' '

    Returns:

    Name Type Description str str

    Flattened, whitespace-normalized text content.

    "},{"location":"omniread/#omniread.HTMLParser.parse_link","title":"parse_link staticmethod","text":"
    parse_link(a: Tag) -> str | None\n

    Extract the hyperlink reference from an <a> element.

    Parameters:

    Name Type Description Default a Tag

    BeautifulSoup tag representing an anchor.

    required

    Returns:

    Type Description str | None

    str | None: The value of the href attribute, or None if absent.

    "},{"location":"omniread/#omniread.HTMLParser.parse_meta","title":"parse_meta","text":"
    parse_meta() -> dict[str, Any]\n

    Extract high-level metadata from the HTML document.

    Returns:

    Type Description dict[str, Any]

    dict[str, Any]: Dictionary containing extracted metadata.

    Notes

    Responsibilities:

    - Extract high-level metadata from the HTML document.\n- This includes: Document title, `<meta>` tag name/property to\n  content mappings.\n
    "},{"location":"omniread/#omniread.HTMLParser.parse_table","title":"parse_table staticmethod","text":"
    parse_table(table: Tag) -> list[list[str]]\n

    Parse an HTML table into a 2D list of strings.

    Parameters:

    Name Type Description Default table Tag

    BeautifulSoup tag representing a <table>.

    required

    Returns:

    Type Description list[list[str]]

    list[list[str]]: A list of rows, where each row is a list of cell text values.

    "},{"location":"omniread/#omniread.HTMLParser.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"omniread/#omniread.HTMLScraper","title":"HTMLScraper","text":"
    HTMLScraper(\n    *,\n    client: httpx.Client | None = None,\n    timeout: float = 15.0,\n    headers: Mapping[str, str] | None = None,\n    follow_redirects: bool = True\n)\n

    Bases: BaseScraper

    Base HTML scraper using httpx.

    Notes

    Responsibilities:

    - This scraper retrieves HTML documents over HTTP(S) and returns\n  them as raw content wrapped in a `Content` object.\n- Fetches raw bytes and metadata only.\n- The scraper uses `httpx.Client` for HTTP requests, enforces an\n  HTML content type, and preserves HTTP response metadata.\n

    Constraints:

    - The scraper does not: Parse HTML, perform retries or backoff,\n  handle non-HTML responses.\n

    Initialize the HTML scraper.

    Parameters:

    Name Type Description Default client Client | None

    Optional pre-configured httpx.Client. If omitted, a client is created internally.

    None timeout float

    Request timeout in seconds.

    15.0 headers Mapping[str, str] | None

    Optional default HTTP headers.

    None follow_redirects bool

    Whether to follow HTTP redirects.

    True"},{"location":"omniread/#omniread.HTMLScraper-functions","title":"Functions","text":""},{"location":"omniread/#omniread.HTMLScraper.fetch","title":"fetch","text":"
    fetch(\n    source: str,\n    *,\n    metadata: Mapping[str, Any] | None = None\n) -> Content\n

    Fetch an HTML document from the given source.

    Parameters:

    Name Type Description Default source str

    URL of the HTML document.

    required metadata Mapping[str, Any] | None

    Optional metadata to be merged into the returned content.

    None

    Returns:

    Name Type Description Content Content

    A Content instance containing raw HTML bytes, source URL, HTML content type, and HTTP response metadata.

    Raises:

    Type Description HTTPError

    If the HTTP request fails.

    ValueError

    If the response is not valid HTML.

    "},{"location":"omniread/#omniread.HTMLScraper.validate_content_type","title":"validate_content_type","text":"
    validate_content_type(response: httpx.Response) -> None\n

    Validate that the HTTP response contains HTML content.

    Parameters:

    Name Type Description Default response Response

    HTTP response returned by httpx.

    required

    Raises:

    Type Description ValueError

    If the Content-Type header is missing or does not indicate HTML content.

    "},{"location":"omniread/#omniread.PDFParser","title":"PDFParser","text":"
    PDFParser(content: Content)\n

    Bases: BaseParser[T], Generic[T]

    Base PDF parser.

    Notes

    Responsibilities:

    - This class enforces PDF content-type compatibility and provides\n  the extension point for implementing concrete PDF parsing strategies.\n

    Constraints:

    - Concrete implementations must define the output type `T` and\n  implement the `parse()` method.\n

    Initialize the parser with content to be parsed.

    Parameters:

    Name Type Description Default content Content

    Content instance to be parsed.

    required

    Raises:

    Type Description ValueError

    If the content type is not supported by this parser.

    "},{"location":"omniread/#omniread.PDFParser-attributes","title":"Attributes","text":""},{"location":"omniread/#omniread.PDFParser.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {PDF}\n

    Set of content types supported by this parser (PDF only).

    "},{"location":"omniread/#omniread.PDFParser-functions","title":"Functions","text":""},{"location":"omniread/#omniread.PDFParser.parse","title":"parse abstractmethod","text":"
    parse() -> T\n

    Parse PDF content into a structured output.

    Returns:

    Name Type Description T T

    Parsed representation of type T.

    Raises:

    Type Description Exception

    Parsing-specific errors as defined by the implementation.

    Notes

    Responsibilities:

    - Implementations must fully interpret the PDF binary payload and\n  return a deterministic, structured output.\n
    "},{"location":"omniread/#omniread.PDFParser.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"omniread/#omniread.PDFScraper","title":"PDFScraper","text":"
    PDFScraper(*, client: BasePDFClient)\n

    Bases: BaseScraper

    Scraper for PDF sources.

    Notes

    Responsibilities:

    - Delegates byte retrieval to a PDF client and normalizes output\n  into `Content`.\n- Preserves caller-provided metadata.\n

    Constraints:

    - The scraper does not perform parsing or interpretation.\n- Does not assume a specific storage backend.\n

    Initialize the PDF scraper.

    Parameters:

    Name Type Description Default client BasePDFClient

    PDF client responsible for retrieving raw PDF bytes.

    required"},{"location":"omniread/#omniread.PDFScraper-functions","title":"Functions","text":""},{"location":"omniread/#omniread.PDFScraper.fetch","title":"fetch","text":"
    fetch(\n    source: Any,\n    *,\n    metadata: Mapping[str, Any] | None = None\n) -> Content\n

    Fetch a PDF document from the given source.

    Parameters:

    Name Type Description Default source Any

    Identifier of the PDF source as understood by the configured PDF client.

    required metadata Mapping[str, Any] | None

    Optional metadata to attach to the returned content.

    None

    Returns:

    Name Type Description Content Content

    A Content instance containing raw PDF bytes, source identifier, PDF content type, and optional metadata.

    Raises:

    Type Description Exception

    Retrieval-specific errors raised by the PDF client.

    "},{"location":"omniread/core/","title":"Core","text":"
    • Content
    • Parser
    • Scraper
    "},{"location":"omniread/core/#omniread.core","title":"omniread.core","text":""},{"location":"omniread/core/#omniread.core--summary","title":"Summary","text":"

    Core domain contracts for OmniRead.

    This package defines the format-agnostic domain layer of OmniRead. It exposes canonical content models and abstract interfaces that are implemented by format-specific modules (HTML, PDF, etc.).

    Public exports from this package are considered stable contracts and are safe for downstream consumers to depend on.

    Submodules:

    • content: Canonical content models and enums.
    • parser: Abstract parsing contracts.
    • scraper: Abstract scraping contracts.

    Format-specific behavior must not be introduced at this layer.

    "},{"location":"omniread/core/#omniread.core--public-api","title":"Public API","text":"
    • Content
    • ContentType
    "},{"location":"omniread/core/#omniread.core-classes","title":"Classes","text":""},{"location":"omniread/core/#omniread.core.BaseParser","title":"BaseParser","text":"
    BaseParser(content: Content)\n

    Bases: ABC, Generic[T]

    Base interface for all parsers.

    Notes

    Guarantees:

    - A parser is a self-contained object that owns the `Content` it is\n  responsible for interpreting.\n- Consumers may rely on early validation of content compatibility\n  and type-stable return values from `parse()`.\n

    Responsibilities:

    - Implementations must declare supported content types via `supported_types`.\n- Implementations must raise parsing-specific exceptions from `parse()`.\n- Implementations must remain deterministic for a given input.\n

    Initialize the parser with content to be parsed.

    Parameters:

    Name Type Description Default content Content

    Content instance to be parsed.

    required

    Raises:

    Type Description ValueError

    If the content type is not supported by this parser.

    "},{"location":"omniread/core/#omniread.core.BaseParser-attributes","title":"Attributes","text":""},{"location":"omniread/core/#omniread.core.BaseParser.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = set()\n

    Set of content types supported by this parser. An empty set indicates that the parser is content-type agnostic.

    "},{"location":"omniread/core/#omniread.core.BaseParser-functions","title":"Functions","text":""},{"location":"omniread/core/#omniread.core.BaseParser.parse","title":"parse abstractmethod","text":"
    parse() -> T\n

    Parse the owned content into structured output.

    Returns:

    Name Type Description T T

    Parsed, structured representation.

    Raises:

    Type Description Exception

    Parsing-specific errors as defined by the implementation.

    Notes

    Responsibilities:

    - Implementations must fully consume the provided content and\n  return a deterministic, structured output.\n
    "},{"location":"omniread/core/#omniread.core.BaseParser.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"omniread/core/#omniread.core.BaseScraper","title":"BaseScraper","text":"

    Bases: ABC

    Base interface for all scrapers.

    Notes

    Responsibilities:

    - A scraper is responsible ONLY for fetching raw content (bytes)\n  from a source. It must not interpret or parse it.\n- A scraper is a stateless acquisition component that retrieves raw\n  content from a source and returns it as a `Content` object.\n- Scrapers define how content is obtained, not what the content means.\n- Implementations may vary in transport mechanism, authentication\n  strategy, retry and backoff behavior.\n

    Constraints:

    - Implementations must not parse content, modify content semantics,\n  or couple scraping logic to a specific parser.\n
    "},{"location":"omniread/core/#omniread.core.BaseScraper-functions","title":"Functions","text":""},{"location":"omniread/core/#omniread.core.BaseScraper.fetch","title":"fetch abstractmethod","text":"
    fetch(\n    source: str,\n    *,\n    metadata: Mapping[str, Any] | None = None\n) -> Content\n

    Fetch raw content from the given source.

    Parameters:

    Name Type Description Default source str

    Location identifier (URL, file path, S3 URI, etc.).

    required metadata Mapping[str, Any] | None

    Optional hints for the scraper (headers, auth, etc.).

    None

    Returns:

    Name Type Description Content Content

    Content object containing raw bytes and metadata.

    Raises:

    Type Description Exception

    Retrieval-specific errors as defined by the implementation.

    Notes

    Responsibilities:

    - Implementations must retrieve the content referenced by `source`\n  and return it as raw bytes wrapped in a `Content` object.\n
    "},{"location":"omniread/core/#omniread.core.Content","title":"Content dataclass","text":"
    Content(\n    raw: bytes,\n    source: str,\n    content_type: ContentType | None = ...,\n    metadata: Mapping[str, Any] | None = ...,\n)\n

    Normalized representation of extracted content.

    Notes

    Responsibilities:

    - A `Content` instance represents a raw content payload along with\n  minimal contextual metadata describing its origin and type.\n- This class is the primary exchange format between scrapers,\n  parsers, and downstream consumers.\n
    "},{"location":"omniread/core/#omniread.core.Content-attributes","title":"Attributes","text":""},{"location":"omniread/core/#omniread.core.Content.content_type","title":"content_type class-attribute instance-attribute","text":"
    content_type: ContentType | None = None\n

    Optional MIME type of the content, if known.

    "},{"location":"omniread/core/#omniread.core.Content.metadata","title":"metadata class-attribute instance-attribute","text":"
    metadata: Mapping[str, Any] | None = None\n

    Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes).

    "},{"location":"omniread/core/#omniread.core.Content.raw","title":"raw instance-attribute","text":"
    raw: bytes\n

    Raw content bytes as retrieved from the source.

    "},{"location":"omniread/core/#omniread.core.Content.source","title":"source instance-attribute","text":"
    source: str\n

    Identifier of the content origin (URL, file path, or logical name).

    "},{"location":"omniread/core/#omniread.core.ContentType","title":"ContentType","text":"

    Bases: str, Enum

    Supported MIME types for extracted content.

    Notes

    Guarantees:

    - This enum represents the declared or inferred media type of the\n  content source.\n- It is primarily used for routing content to the appropriate\n  parser or downstream consumer.\n
    "},{"location":"omniread/core/#omniread.core.ContentType-attributes","title":"Attributes","text":""},{"location":"omniread/core/#omniread.core.ContentType.CSV","title":"CSV class-attribute instance-attribute","text":"
    CSV = 'text/csv'\n

    Comma-separated-value document content.

    "},{"location":"omniread/core/#omniread.core.ContentType.HTML","title":"HTML class-attribute instance-attribute","text":"
    HTML = 'text/html'\n

    HTML document content.

    "},{"location":"omniread/core/#omniread.core.ContentType.JSON","title":"JSON class-attribute instance-attribute","text":"
    JSON = 'application/json'\n

    JSON document content.

    "},{"location":"omniread/core/#omniread.core.ContentType.PDF","title":"PDF class-attribute instance-attribute","text":"
    PDF = 'application/pdf'\n

    PDF document content.

    "},{"location":"omniread/core/#omniread.core.ContentType.XLSX","title":"XLSX class-attribute instance-attribute","text":"
    XLSX = \"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet\"\n

    Office Open XML spreadsheet (xlsx/xlsm) content.

    "},{"location":"omniread/core/#omniread.core.ContentType.XML","title":"XML class-attribute instance-attribute","text":"
    XML = 'application/xml'\n

    XML document content.

    "},{"location":"omniread/core/content/","title":"Content","text":""},{"location":"omniread/core/content/#omniread.core.content","title":"omniread.core.content","text":""},{"location":"omniread/core/content/#omniread.core.content--summary","title":"Summary","text":"

    Canonical content models for OmniRead.

    This module defines the format-agnostic content representation used across all parsers and scrapers in OmniRead.

    The models defined here represent what was extracted, not how it was retrieved or parsed. Format-specific behavior and metadata must not alter the semantic meaning of these models.

    "},{"location":"omniread/core/content/#omniread.core.content-classes","title":"Classes","text":""},{"location":"omniread/core/content/#omniread.core.content.Content","title":"Content dataclass","text":"
    Content(\n    raw: bytes,\n    source: str,\n    content_type: ContentType | None = ...,\n    metadata: Mapping[str, Any] | None = ...,\n)\n

    Normalized representation of extracted content.

    Notes

    Responsibilities:

    - A `Content` instance represents a raw content payload along with\n  minimal contextual metadata describing its origin and type.\n- This class is the primary exchange format between scrapers,\n  parsers, and downstream consumers.\n
    "},{"location":"omniread/core/content/#omniread.core.content.Content-attributes","title":"Attributes","text":""},{"location":"omniread/core/content/#omniread.core.content.Content.content_type","title":"content_type class-attribute instance-attribute","text":"
    content_type: ContentType | None = None\n

    Optional MIME type of the content, if known.

    "},{"location":"omniread/core/content/#omniread.core.content.Content.metadata","title":"metadata class-attribute instance-attribute","text":"
    metadata: Mapping[str, Any] | None = None\n

    Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes).

    "},{"location":"omniread/core/content/#omniread.core.content.Content.raw","title":"raw instance-attribute","text":"
    raw: bytes\n

    Raw content bytes as retrieved from the source.

    "},{"location":"omniread/core/content/#omniread.core.content.Content.source","title":"source instance-attribute","text":"
    source: str\n

    Identifier of the content origin (URL, file path, or logical name).

    "},{"location":"omniread/core/content/#omniread.core.content.ContentType","title":"ContentType","text":"

    Bases: str, Enum

    Supported MIME types for extracted content.

    Notes

    Guarantees:

    - This enum represents the declared or inferred media type of the\n  content source.\n- It is primarily used for routing content to the appropriate\n  parser or downstream consumer.\n
    "},{"location":"omniread/core/content/#omniread.core.content.ContentType-attributes","title":"Attributes","text":""},{"location":"omniread/core/content/#omniread.core.content.ContentType.CSV","title":"CSV class-attribute instance-attribute","text":"
    CSV = 'text/csv'\n

    Comma-separated-value document content.

    "},{"location":"omniread/core/content/#omniread.core.content.ContentType.HTML","title":"HTML class-attribute instance-attribute","text":"
    HTML = 'text/html'\n

    HTML document content.

    "},{"location":"omniread/core/content/#omniread.core.content.ContentType.JSON","title":"JSON class-attribute instance-attribute","text":"
    JSON = 'application/json'\n

    JSON document content.

    "},{"location":"omniread/core/content/#omniread.core.content.ContentType.PDF","title":"PDF class-attribute instance-attribute","text":"
    PDF = 'application/pdf'\n

    PDF document content.

    "},{"location":"omniread/core/content/#omniread.core.content.ContentType.XLSX","title":"XLSX class-attribute instance-attribute","text":"
    XLSX = \"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet\"\n

    Office Open XML spreadsheet (xlsx/xlsm) content.

    "},{"location":"omniread/core/content/#omniread.core.content.ContentType.XML","title":"XML class-attribute instance-attribute","text":"
    XML = 'application/xml'\n

    XML document content.

    "},{"location":"omniread/core/parser/","title":"Parser","text":""},{"location":"omniread/core/parser/#omniread.core.parser","title":"omniread.core.parser","text":""},{"location":"omniread/core/parser/#omniread.core.parser--summary","title":"Summary","text":"

    Abstract parsing contracts for OmniRead.

    This module defines the format-agnostic parser interface used to transform raw content into structured, typed representations.

    Parsers are responsible for:

    • Interpreting a single Content instance
    • Validating compatibility with the content type
    • Producing a structured output suitable for downstream consumers

    Parsers are not responsible for:

    • Fetching or acquiring content
    • Performing retries or error recovery
    • Managing multiple content sources
    "},{"location":"omniread/core/parser/#omniread.core.parser-classes","title":"Classes","text":""},{"location":"omniread/core/parser/#omniread.core.parser.BaseParser","title":"BaseParser","text":"
    BaseParser(content: Content)\n

    Bases: ABC, Generic[T]

    Base interface for all parsers.

    Notes

    Guarantees:

    - A parser is a self-contained object that owns the `Content` it is\n  responsible for interpreting.\n- Consumers may rely on early validation of content compatibility\n  and type-stable return values from `parse()`.\n

    Responsibilities:

    - Implementations must declare supported content types via `supported_types`.\n- Implementations must raise parsing-specific exceptions from `parse()`.\n- Implementations must remain deterministic for a given input.\n

    Initialize the parser with content to be parsed.

    Parameters:

    Name Type Description Default content Content

    Content instance to be parsed.

    required

    Raises:

    Type Description ValueError

    If the content type is not supported by this parser.

    "},{"location":"omniread/core/parser/#omniread.core.parser.BaseParser-attributes","title":"Attributes","text":""},{"location":"omniread/core/parser/#omniread.core.parser.BaseParser.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = set()\n

    Set of content types supported by this parser. An empty set indicates that the parser is content-type agnostic.

    "},{"location":"omniread/core/parser/#omniread.core.parser.BaseParser-functions","title":"Functions","text":""},{"location":"omniread/core/parser/#omniread.core.parser.BaseParser.parse","title":"parse abstractmethod","text":"
    parse() -> T\n

    Parse the owned content into structured output.

    Returns:

    Name Type Description T T

    Parsed, structured representation.

    Raises:

    Type Description Exception

    Parsing-specific errors as defined by the implementation.

    Notes

    Responsibilities:

    - Implementations must fully consume the provided content and\n  return a deterministic, structured output.\n
    "},{"location":"omniread/core/parser/#omniread.core.parser.BaseParser.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"omniread/core/scraper/","title":"Scraper","text":""},{"location":"omniread/core/scraper/#omniread.core.scraper","title":"omniread.core.scraper","text":""},{"location":"omniread/core/scraper/#omniread.core.scraper--summary","title":"Summary","text":"

    Abstract scraping contracts for OmniRead.

    This module defines the format-agnostic scraper interface responsible for acquiring raw content from external sources.

    Scrapers are responsible for:

    • Locating and retrieving raw content bytes
    • Attaching minimal contextual metadata
    • Returning normalized Content objects

    Scrapers are explicitly NOT responsible for:

    • Parsing or interpreting content
    • Inferring structure or semantics
    • Performing content-type specific processing

    All interpretation must be delegated to parsers.

    "},{"location":"omniread/core/scraper/#omniread.core.scraper-classes","title":"Classes","text":""},{"location":"omniread/core/scraper/#omniread.core.scraper.BaseScraper","title":"BaseScraper","text":"

    Bases: ABC

    Base interface for all scrapers.

    Notes

    Responsibilities:

    - A scraper is responsible ONLY for fetching raw content (bytes)\n  from a source. It must not interpret or parse it.\n- A scraper is a stateless acquisition component that retrieves raw\n  content from a source and returns it as a `Content` object.\n- Scrapers define how content is obtained, not what the content means.\n- Implementations may vary in transport mechanism, authentication\n  strategy, retry and backoff behavior.\n

    Constraints:

    - Implementations must not parse content, modify content semantics,\n  or couple scraping logic to a specific parser.\n
    "},{"location":"omniread/core/scraper/#omniread.core.scraper.BaseScraper-functions","title":"Functions","text":""},{"location":"omniread/core/scraper/#omniread.core.scraper.BaseScraper.fetch","title":"fetch abstractmethod","text":"
    fetch(\n    source: str,\n    *,\n    metadata: Mapping[str, Any] | None = None\n) -> Content\n

    Fetch raw content from the given source.

    Parameters:

    Name Type Description Default source str

    Location identifier (URL, file path, S3 URI, etc.).

    required metadata Mapping[str, Any] | None

    Optional hints for the scraper (headers, auth, etc.).

    None

    Returns:

    Name Type Description Content Content

    Content object containing raw bytes and metadata.

    Raises:

    Type Description Exception

    Retrieval-specific errors as defined by the implementation.

    Notes

    Responsibilities:

    - Implementations must retrieve the content referenced by `source`\n  and return it as raw bytes wrapped in a `Content` object.\n
    "},{"location":"omniread/csv/","title":"Csv","text":"
    • Client
    • Parser
    • Parser Base
    • Scraper
    "},{"location":"omniread/csv/#omniread.csv","title":"omniread.csv","text":""},{"location":"omniread/csv/#omniread.csv--summary","title":"Summary","text":"

    CSV subpackage for OmniRead.

    Provides acquisition and parsing of comma-separated-value content:

    • BaseCsvClient: abstract backing-store client for csv bytes.
    • FileSystemCsvClient: local filesystem implementation.
    • CsvScraper: wraps fetched bytes into canonical Content.
    • CsvParserBase: content-type-enforcing parser contract.
    • CsvParser: generic string-row parser built on the standard csv module.
    "},{"location":"omniread/csv/#omniread.csv-classes","title":"Classes","text":""},{"location":"omniread/csv/#omniread.csv.BaseCsvClient","title":"BaseCsvClient","text":"

    Bases: ABC

    Abstract client responsible for retrieving csv bytes.

    Retrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).

    Notes

    Responsibilities:

    - Implementations must accept a source identifier appropriate to\n  the backing store.\n- Return the full csv binary payload.\n- Raise retrieval-specific errors on failure.\n
    "},{"location":"omniread/csv/#omniread.csv.BaseCsvClient-functions","title":"Functions","text":""},{"location":"omniread/csv/#omniread.csv.BaseCsvClient.fetch","title":"fetch abstractmethod","text":"
    fetch(source: Any) -> bytes\n

    Fetch raw csv bytes from the given source.

    Parameters:

    Name Type Description Default source Any

    Identifier of the csv location, such as a file path, object storage key, or remote reference.

    required

    Returns:

    Name Type Description bytes bytes

    Raw csv bytes.

    Raises:

    Type Description Exception

    Retrieval-specific errors defined by the implementation.

    "},{"location":"omniread/csv/#omniread.csv.CsvParser","title":"CsvParser","text":"
    CsvParser(content: Content)\n

    Bases: CsvParserBase[list[list[str]]]

    Generic csv parser producing string rows from the document.

    Notes

    Responsibilities:

    - Decode the payload (UTF-8 with BOM support, Latin-1 fallback).\n- Detect the delimiter from a leading sample (`,` `;` tab `|`),\n  defaulting to `,`.\n- Normalize cells into deterministic stripped string values.\n- Expose row extraction helpers mirroring `XlsxParser.rows`.\n

    Constraints:

    - All values are strings; consumers requiring typed values must\n  convert on their side.\n- Quoted fields containing delimiters/newlines are handled by\n  the standard ``csv`` module.\n

    Initialize the parser.

    Parameters:

    Name Type Description Default content Content

    CSV content to parse; its type must be supported.

    required"},{"location":"omniread/csv/#omniread.csv.CsvParser-attributes","title":"Attributes","text":""},{"location":"omniread/csv/#omniread.csv.CsvParser.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {CSV}\n

    Set of content types supported by this parser (CSV only).

    "},{"location":"omniread/csv/#omniread.csv.CsvParser-functions","title":"Functions","text":""},{"location":"omniread/csv/#omniread.csv.CsvParser.parse","title":"parse","text":"
    parse() -> list[list[str]]\n

    Parse the document into normalized string rows.

    Returns:

    Type Description list[list[str]]

    list[list[str]]: Rows of the document.

    "},{"location":"omniread/csv/#omniread.csv.CsvParser.rows","title":"rows","text":"
    rows(*, skip_empty: bool = True) -> list[list[str]]\n

    Extract normalized string rows from the document.

    Parameters:

    Name Type Description Default skip_empty bool

    When True (default), rows whose cells are all blank are omitted.

    True

    Returns:

    Type Description list[list[str]]

    list[list[str]]: Normalized rows; trailing blank cells are trimmed per row.

    "},{"location":"omniread/csv/#omniread.csv.CsvParser.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"omniread/csv/#omniread.csv.CsvParserBase","title":"CsvParserBase","text":"
    CsvParserBase(content: Content)\n

    Bases: BaseParser[T], Generic[T]

    Base csv parser.

    Notes

    Responsibilities:

    - This class enforces csv content-type compatibility and provides\n  the extension point for implementing concrete csv parsing\n  strategies.\n

    Constraints:

    - Concrete implementations must define the output type `T` and\n  implement the `parse()` method.\n

    Initialize the parser with content to be parsed.

    Parameters:

    Name Type Description Default content Content

    Content instance to be parsed.

    required

    Raises:

    Type Description ValueError

    If the content type is not supported by this parser.

    "},{"location":"omniread/csv/#omniread.csv.CsvParserBase-attributes","title":"Attributes","text":""},{"location":"omniread/csv/#omniread.csv.CsvParserBase.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {CSV}\n

    Set of content types supported by this parser (CSV only).

    "},{"location":"omniread/csv/#omniread.csv.CsvParserBase-functions","title":"Functions","text":""},{"location":"omniread/csv/#omniread.csv.CsvParserBase.parse","title":"parse abstractmethod","text":"
    parse() -> T\n

    Parse csv content into a structured output.

    Returns:

    Name Type Description T T

    Parsed representation of type T.

    Raises:

    Type Description Exception

    Parsing-specific errors as defined by the implementation.

    "},{"location":"omniread/csv/#omniread.csv.CsvParserBase.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"omniread/csv/#omniread.csv.CsvScraper","title":"CsvScraper","text":"
    CsvScraper(*, client: BaseCsvClient)\n

    Scraper for csv documents.

    Notes

    Responsibilities:

    - Fetch raw csv bytes via the configured client.\n- Wrap the payload in a canonical `Content` instance with the\n  CSV content type and source identifier.\n

    Constraints:

    - The scraper does not perform parsing or interpretation.\n- Does not assume a specific storage backend.\n

    Initialize the CSV scraper.

    Parameters:

    Name Type Description Default client BaseCsvClient

    Client responsible for retrieving raw csv bytes.

    required"},{"location":"omniread/csv/#omniread.csv.CsvScraper-functions","title":"Functions","text":""},{"location":"omniread/csv/#omniread.csv.CsvScraper.fetch","title":"fetch","text":"
    fetch(\n    source: Any,\n    *,\n    metadata: Mapping[str, Any] | None = None\n) -> Content\n

    Fetch a csv document from the given source.

    Parameters:

    Name Type Description Default source Any

    Identifier of the csv source as understood by the configured client.

    required metadata Mapping[str, Any] | None

    Optional metadata to attach to the returned content.

    None

    Returns:

    Name Type Description Content Content

    A Content instance containing raw csv bytes, source identifier, CSV content type, and optional metadata.

    Raises:

    Type Description Exception

    Retrieval-specific errors raised by the client.

    "},{"location":"omniread/csv/#omniread.csv.FileSystemCsvClient","title":"FileSystemCsvClient","text":"

    Bases: BaseCsvClient

    CSV client that reads from the local filesystem.

    Notes

    Guarantees:

    - This client reads csv files directly from the disk and\n  returns their raw binary contents.\n
    "},{"location":"omniread/csv/#omniread.csv.FileSystemCsvClient-functions","title":"Functions","text":""},{"location":"omniread/csv/#omniread.csv.FileSystemCsvClient.fetch","title":"fetch","text":"
    fetch(path: Path) -> bytes\n

    Read a csv file from the local filesystem.

    Parameters:

    Name Type Description Default path Path

    Filesystem path to the csv file.

    required

    Returns:

    Name Type Description bytes bytes

    Raw csv bytes.

    Raises:

    Type Description FileNotFoundError

    If the path does not exist.

    ValueError

    If the path exists but is not a file.

    "},{"location":"omniread/csv/client/","title":"Client","text":""},{"location":"omniread/csv/client/#omniread.csv.client","title":"omniread.csv.client","text":""},{"location":"omniread/csv/client/#omniread.csv.client--summary","title":"Summary","text":"

    CSV client abstractions for OmniRead.

    This module defines the client layer responsible for retrieving raw comma-separated-value document bytes from a concrete backing store.

    Clients provide low-level access to csv binaries and are intentionally decoupled from scraping and parsing logic. They do not perform validation, interpretation, or content extraction.

    Typical backing stores include:

    • Local filesystems
    • Object storage (S3, GCS, etc.)
    • Network file systems
    "},{"location":"omniread/csv/client/#omniread.csv.client-classes","title":"Classes","text":""},{"location":"omniread/csv/client/#omniread.csv.client.BaseCsvClient","title":"BaseCsvClient","text":"

    Bases: ABC

    Abstract client responsible for retrieving csv bytes.

    Retrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).

    Notes

    Responsibilities:

    - Implementations must accept a source identifier appropriate to\n  the backing store.\n- Return the full csv binary payload.\n- Raise retrieval-specific errors on failure.\n
    "},{"location":"omniread/csv/client/#omniread.csv.client.BaseCsvClient-functions","title":"Functions","text":""},{"location":"omniread/csv/client/#omniread.csv.client.BaseCsvClient.fetch","title":"fetch abstractmethod","text":"
    fetch(source: Any) -> bytes\n

    Fetch raw csv bytes from the given source.

    Parameters:

    Name Type Description Default source Any

    Identifier of the csv location, such as a file path, object storage key, or remote reference.

    required

    Returns:

    Name Type Description bytes bytes

    Raw csv bytes.

    Raises:

    Type Description Exception

    Retrieval-specific errors defined by the implementation.

    "},{"location":"omniread/csv/client/#omniread.csv.client.FileSystemCsvClient","title":"FileSystemCsvClient","text":"

    Bases: BaseCsvClient

    CSV client that reads from the local filesystem.

    Notes

    Guarantees:

    - This client reads csv files directly from the disk and\n  returns their raw binary contents.\n
    "},{"location":"omniread/csv/client/#omniread.csv.client.FileSystemCsvClient-functions","title":"Functions","text":""},{"location":"omniread/csv/client/#omniread.csv.client.FileSystemCsvClient.fetch","title":"fetch","text":"
    fetch(path: Path) -> bytes\n

    Read a csv file from the local filesystem.

    Parameters:

    Name Type Description Default path Path

    Filesystem path to the csv file.

    required

    Returns:

    Name Type Description bytes bytes

    Raw csv bytes.

    Raises:

    Type Description FileNotFoundError

    If the path does not exist.

    ValueError

    If the path exists but is not a file.

    "},{"location":"omniread/csv/parser/","title":"Parser","text":""},{"location":"omniread/csv/parser/#omniread.csv.parser","title":"omniread.csv.parser","text":""},{"location":"omniread/csv/parser/#omniread.csv.parser--summary","title":"Summary","text":"

    CSV parser implementations for OmniRead.

    This module provides a concrete, generic parser for comma-separated-value documents. It exposes records as lists of string cells so downstream consumers can interpret tabular content without depending on the csv module directly.

    The parser is intentionally statement-agnostic: it performs no header detection or column interpretation beyond basic cell normalization and delimiter detection.

    "},{"location":"omniread/csv/parser/#omniread.csv.parser-classes","title":"Classes","text":""},{"location":"omniread/csv/parser/#omniread.csv.parser.CsvParser","title":"CsvParser","text":"
    CsvParser(content: Content)\n

    Bases: CsvParserBase[list[list[str]]]

    Generic csv parser producing string rows from the document.

    Notes

    Responsibilities:

    - Decode the payload (UTF-8 with BOM support, Latin-1 fallback).\n- Detect the delimiter from a leading sample (`,` `;` tab `|`),\n  defaulting to `,`.\n- Normalize cells into deterministic stripped string values.\n- Expose row extraction helpers mirroring `XlsxParser.rows`.\n

    Constraints:

    - All values are strings; consumers requiring typed values must\n  convert on their side.\n- Quoted fields containing delimiters/newlines are handled by\n  the standard ``csv`` module.\n

    Initialize the parser.

    Parameters:

    Name Type Description Default content Content

    CSV content to parse; its type must be supported.

    required"},{"location":"omniread/csv/parser/#omniread.csv.parser.CsvParser-attributes","title":"Attributes","text":""},{"location":"omniread/csv/parser/#omniread.csv.parser.CsvParser.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {CSV}\n

    Set of content types supported by this parser (CSV only).

    "},{"location":"omniread/csv/parser/#omniread.csv.parser.CsvParser-functions","title":"Functions","text":""},{"location":"omniread/csv/parser/#omniread.csv.parser.CsvParser.parse","title":"parse","text":"
    parse() -> list[list[str]]\n

    Parse the document into normalized string rows.

    Returns:

    Type Description list[list[str]]

    list[list[str]]: Rows of the document.

    "},{"location":"omniread/csv/parser/#omniread.csv.parser.CsvParser.rows","title":"rows","text":"
    rows(*, skip_empty: bool = True) -> list[list[str]]\n

    Extract normalized string rows from the document.

    Parameters:

    Name Type Description Default skip_empty bool

    When True (default), rows whose cells are all blank are omitted.

    True

    Returns:

    Type Description list[list[str]]

    list[list[str]]: Normalized rows; trailing blank cells are trimmed per row.

    "},{"location":"omniread/csv/parser/#omniread.csv.parser.CsvParser.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"omniread/csv/parser_base/","title":"Parser Base","text":""},{"location":"omniread/csv/parser_base/#omniread.csv.parser_base","title":"omniread.csv.parser_base","text":""},{"location":"omniread/csv/parser_base/#omniread.csv.parser_base--summary","title":"Summary","text":"

    CSV parser base implementation for OmniRead.

    This module defines the CSV-specific parser contract, extending the format-agnostic BaseParser with constraints appropriate for comma-separated-value documents.

    "},{"location":"omniread/csv/parser_base/#omniread.csv.parser_base-classes","title":"Classes","text":""},{"location":"omniread/csv/parser_base/#omniread.csv.parser_base.CsvParserBase","title":"CsvParserBase","text":"
    CsvParserBase(content: Content)\n

    Bases: BaseParser[T], Generic[T]

    Base csv parser.

    Notes

    Responsibilities:

    - This class enforces csv content-type compatibility and provides\n  the extension point for implementing concrete csv parsing\n  strategies.\n

    Constraints:

    - Concrete implementations must define the output type `T` and\n  implement the `parse()` method.\n

    Initialize the parser with content to be parsed.

    Parameters:

    Name Type Description Default content Content

    Content instance to be parsed.

    required

    Raises:

    Type Description ValueError

    If the content type is not supported by this parser.

    "},{"location":"omniread/csv/parser_base/#omniread.csv.parser_base.CsvParserBase-attributes","title":"Attributes","text":""},{"location":"omniread/csv/parser_base/#omniread.csv.parser_base.CsvParserBase.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {CSV}\n

    Set of content types supported by this parser (CSV only).

    "},{"location":"omniread/csv/parser_base/#omniread.csv.parser_base.CsvParserBase-functions","title":"Functions","text":""},{"location":"omniread/csv/parser_base/#omniread.csv.parser_base.CsvParserBase.parse","title":"parse abstractmethod","text":"
    parse() -> T\n

    Parse csv content into a structured output.

    Returns:

    Name Type Description T T

    Parsed representation of type T.

    Raises:

    Type Description Exception

    Parsing-specific errors as defined by the implementation.

    "},{"location":"omniread/csv/parser_base/#omniread.csv.parser_base.CsvParserBase.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"omniread/csv/scraper/","title":"Scraper","text":""},{"location":"omniread/csv/scraper/#omniread.csv.scraper","title":"omniread.csv.scraper","text":""},{"location":"omniread/csv/scraper/#omniread.csv.scraper--summary","title":"Summary","text":"

    CSV scraper for OmniRead.

    This module defines the scraper responsible for acquiring raw comma-separated-value document content from a backing store via a configured client.

    The scraper does not interpret or parse the acquired bytes; it wraps them in the canonical Content model.

    "},{"location":"omniread/csv/scraper/#omniread.csv.scraper-classes","title":"Classes","text":""},{"location":"omniread/csv/scraper/#omniread.csv.scraper.CsvScraper","title":"CsvScraper","text":"
    CsvScraper(*, client: BaseCsvClient)\n

    Scraper for csv documents.

    Notes

    Responsibilities:

    - Fetch raw csv bytes via the configured client.\n- Wrap the payload in a canonical `Content` instance with the\n  CSV content type and source identifier.\n

    Constraints:

    - The scraper does not perform parsing or interpretation.\n- Does not assume a specific storage backend.\n

    Initialize the CSV scraper.

    Parameters:

    Name Type Description Default client BaseCsvClient

    Client responsible for retrieving raw csv bytes.

    required"},{"location":"omniread/csv/scraper/#omniread.csv.scraper.CsvScraper-functions","title":"Functions","text":""},{"location":"omniread/csv/scraper/#omniread.csv.scraper.CsvScraper.fetch","title":"fetch","text":"
    fetch(\n    source: Any,\n    *,\n    metadata: Mapping[str, Any] | None = None\n) -> Content\n

    Fetch a csv document from the given source.

    Parameters:

    Name Type Description Default source Any

    Identifier of the csv source as understood by the configured client.

    required metadata Mapping[str, Any] | None

    Optional metadata to attach to the returned content.

    None

    Returns:

    Name Type Description Content Content

    A Content instance containing raw csv bytes, source identifier, CSV content type, and optional metadata.

    Raises:

    Type Description Exception

    Retrieval-specific errors raised by the client.

    "},{"location":"omniread/html/","title":"Html","text":"
    • Parser
    • Scraper
    "},{"location":"omniread/html/#omniread.html","title":"omniread.html","text":""},{"location":"omniread/html/#omniread.html--summary","title":"Summary","text":"

    HTML format implementation for OmniRead.

    This package provides HTML-specific implementations of the core OmniRead contracts defined in omniread.core.

    It includes:

    • HTML parsers that interpret HTML content.
    • HTML scrapers that retrieve HTML documents.

    Key characteristics:

    • Implements, but does not redefine, core contracts.
    • May contain HTML-specific behavior and edge-case handling.
    • Produces canonical content models defined in omniread.core.content.

    Consumers should depend on omniread.core interfaces wherever possible and use this package only when HTML-specific behavior is required.

    "},{"location":"omniread/html/#omniread.html--public-api","title":"Public API","text":"
    • HTMLScraper
    • HTMLParser
    "},{"location":"omniread/html/#omniread.html-classes","title":"Classes","text":""},{"location":"omniread/html/#omniread.html.HTMLParser","title":"HTMLParser","text":"
    HTMLParser(content: Content, features: str = 'html.parser')\n

    Bases: BaseParser[T], Generic[T]

    Base HTML parser.

    Notes

    Responsibilities:

    - This class extends the core `BaseParser` with HTML-specific behavior,\n  including DOM parsing via BeautifulSoup and reusable extraction helpers.\n- Provides reusable helpers for HTML extraction. Concrete parsers must\n  explicitly define the return type.\n

    Guarantees:

    - Accepts only HTML content.\n- Owns a parsed BeautifulSoup DOM tree.\n- Provides pure helper utilities for common HTML structures.\n

    Constraints:

    - Concrete subclasses must define the output type `T` and implement\n  the `parse()` method.\n

    Initialize the HTML parser.

    Parameters:

    Name Type Description Default content Content

    HTML content to be parsed.

    required features str

    BeautifulSoup parser backend to use (e.g., 'html.parser', 'lxml').

    'html.parser'

    Raises:

    Type Description ValueError

    If the content is empty or not valid HTML.

    "},{"location":"omniread/html/#omniread.html.HTMLParser-attributes","title":"Attributes","text":""},{"location":"omniread/html/#omniread.html.HTMLParser.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {HTML}\n

    Set of content types supported by this parser (HTML only).

    "},{"location":"omniread/html/#omniread.html.HTMLParser-functions","title":"Functions","text":""},{"location":"omniread/html/#omniread.html.HTMLParser.parse","title":"parse abstractmethod","text":"
    parse() -> T\n

    Fully parse the HTML content into structured output.

    Returns:

    Name Type Description T T

    Parsed representation of type T.

    Notes

    Responsibilities:

    - Implementations must fully interpret the HTML DOM and return a\n  deterministic, structured output.\n
    "},{"location":"omniread/html/#omniread.html.HTMLParser.parse_div","title":"parse_div staticmethod","text":"
    parse_div(div: Tag, *, separator: str = ' ') -> str\n

    Extract normalized text from a <div> element.

    Parameters:

    Name Type Description Default div Tag

    BeautifulSoup tag representing a <div>.

    required separator str

    String used to separate text nodes.

    ' '

    Returns:

    Name Type Description str str

    Flattened, whitespace-normalized text content.

    "},{"location":"omniread/html/#omniread.html.HTMLParser.parse_link","title":"parse_link staticmethod","text":"
    parse_link(a: Tag) -> str | None\n

    Extract the hyperlink reference from an <a> element.

    Parameters:

    Name Type Description Default a Tag

    BeautifulSoup tag representing an anchor.

    required

    Returns:

    Type Description str | None

    str | None: The value of the href attribute, or None if absent.

    "},{"location":"omniread/html/#omniread.html.HTMLParser.parse_meta","title":"parse_meta","text":"
    parse_meta() -> dict[str, Any]\n

    Extract high-level metadata from the HTML document.

    Returns:

    Type Description dict[str, Any]

    dict[str, Any]: Dictionary containing extracted metadata.

    Notes

    Responsibilities:

    - Extract high-level metadata from the HTML document.\n- This includes: Document title, `<meta>` tag name/property to\n  content mappings.\n
    "},{"location":"omniread/html/#omniread.html.HTMLParser.parse_table","title":"parse_table staticmethod","text":"
    parse_table(table: Tag) -> list[list[str]]\n

    Parse an HTML table into a 2D list of strings.

    Parameters:

    Name Type Description Default table Tag

    BeautifulSoup tag representing a <table>.

    required

    Returns:

    Type Description list[list[str]]

    list[list[str]]: A list of rows, where each row is a list of cell text values.

    "},{"location":"omniread/html/#omniread.html.HTMLParser.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"omniread/html/#omniread.html.HTMLScraper","title":"HTMLScraper","text":"
    HTMLScraper(\n    *,\n    client: httpx.Client | None = None,\n    timeout: float = 15.0,\n    headers: Mapping[str, str] | None = None,\n    follow_redirects: bool = True\n)\n

    Bases: BaseScraper

    Base HTML scraper using httpx.

    Notes

    Responsibilities:

    - This scraper retrieves HTML documents over HTTP(S) and returns\n  them as raw content wrapped in a `Content` object.\n- Fetches raw bytes and metadata only.\n- The scraper uses `httpx.Client` for HTTP requests, enforces an\n  HTML content type, and preserves HTTP response metadata.\n

    Constraints:

    - The scraper does not: Parse HTML, perform retries or backoff,\n  handle non-HTML responses.\n

    Initialize the HTML scraper.

    Parameters:

    Name Type Description Default client Client | None

    Optional pre-configured httpx.Client. If omitted, a client is created internally.

    None timeout float

    Request timeout in seconds.

    15.0 headers Mapping[str, str] | None

    Optional default HTTP headers.

    None follow_redirects bool

    Whether to follow HTTP redirects.

    True"},{"location":"omniread/html/#omniread.html.HTMLScraper-functions","title":"Functions","text":""},{"location":"omniread/html/#omniread.html.HTMLScraper.fetch","title":"fetch","text":"
    fetch(\n    source: str,\n    *,\n    metadata: Mapping[str, Any] | None = None\n) -> Content\n

    Fetch an HTML document from the given source.

    Parameters:

    Name Type Description Default source str

    URL of the HTML document.

    required metadata Mapping[str, Any] | None

    Optional metadata to be merged into the returned content.

    None

    Returns:

    Name Type Description Content Content

    A Content instance containing raw HTML bytes, source URL, HTML content type, and HTTP response metadata.

    Raises:

    Type Description HTTPError

    If the HTTP request fails.

    ValueError

    If the response is not valid HTML.

    "},{"location":"omniread/html/#omniread.html.HTMLScraper.validate_content_type","title":"validate_content_type","text":"
    validate_content_type(response: httpx.Response) -> None\n

    Validate that the HTTP response contains HTML content.

    Parameters:

    Name Type Description Default response Response

    HTTP response returned by httpx.

    required

    Raises:

    Type Description ValueError

    If the Content-Type header is missing or does not indicate HTML content.

    "},{"location":"omniread/html/parser/","title":"Parser","text":""},{"location":"omniread/html/parser/#omniread.html.parser","title":"omniread.html.parser","text":""},{"location":"omniread/html/parser/#omniread.html.parser--summary","title":"Summary","text":"

    HTML parser base implementations for OmniRead.

    This module provides reusable HTML parsing utilities built on top of the abstract parser contracts defined in omniread.core.parser.

    It supplies:

    • Content-type enforcement for HTML inputs
    • BeautifulSoup initialization and lifecycle management
    • Common helper methods for extracting structured data from HTML elements

    Concrete parsers must subclass HTMLParser and implement the parse() method to return a structured representation appropriate for their use case.

    "},{"location":"omniread/html/parser/#omniread.html.parser-classes","title":"Classes","text":""},{"location":"omniread/html/parser/#omniread.html.parser.HTMLParser","title":"HTMLParser","text":"
    HTMLParser(content: Content, features: str = 'html.parser')\n

    Bases: BaseParser[T], Generic[T]

    Base HTML parser.

    Notes

    Responsibilities:

    - This class extends the core `BaseParser` with HTML-specific behavior,\n  including DOM parsing via BeautifulSoup and reusable extraction helpers.\n- Provides reusable helpers for HTML extraction. Concrete parsers must\n  explicitly define the return type.\n

    Guarantees:

    - Accepts only HTML content.\n- Owns a parsed BeautifulSoup DOM tree.\n- Provides pure helper utilities for common HTML structures.\n

    Constraints:

    - Concrete subclasses must define the output type `T` and implement\n  the `parse()` method.\n

    Initialize the HTML parser.

    Parameters:

    Name Type Description Default content Content

    HTML content to be parsed.

    required features str

    BeautifulSoup parser backend to use (e.g., 'html.parser', 'lxml').

    'html.parser'

    Raises:

    Type Description ValueError

    If the content is empty or not valid HTML.

    "},{"location":"omniread/html/parser/#omniread.html.parser.HTMLParser-attributes","title":"Attributes","text":""},{"location":"omniread/html/parser/#omniread.html.parser.HTMLParser.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {HTML}\n

    Set of content types supported by this parser (HTML only).

    "},{"location":"omniread/html/parser/#omniread.html.parser.HTMLParser-functions","title":"Functions","text":""},{"location":"omniread/html/parser/#omniread.html.parser.HTMLParser.parse","title":"parse abstractmethod","text":"
    parse() -> T\n

    Fully parse the HTML content into structured output.

    Returns:

    Name Type Description T T

    Parsed representation of type T.

    Notes

    Responsibilities:

    - Implementations must fully interpret the HTML DOM and return a\n  deterministic, structured output.\n
    "},{"location":"omniread/html/parser/#omniread.html.parser.HTMLParser.parse_div","title":"parse_div staticmethod","text":"
    parse_div(div: Tag, *, separator: str = ' ') -> str\n

    Extract normalized text from a <div> element.

    Parameters:

    Name Type Description Default div Tag

    BeautifulSoup tag representing a <div>.

    required separator str

    String used to separate text nodes.

    ' '

    Returns:

    Name Type Description str str

    Flattened, whitespace-normalized text content.

    "},{"location":"omniread/html/parser/#omniread.html.parser.HTMLParser.parse_link","title":"parse_link staticmethod","text":"
    parse_link(a: Tag) -> str | None\n

    Extract the hyperlink reference from an <a> element.

    Parameters:

    Name Type Description Default a Tag

    BeautifulSoup tag representing an anchor.

    required

    Returns:

    Type Description str | None

    str | None: The value of the href attribute, or None if absent.

    "},{"location":"omniread/html/parser/#omniread.html.parser.HTMLParser.parse_meta","title":"parse_meta","text":"
    parse_meta() -> dict[str, Any]\n

    Extract high-level metadata from the HTML document.

    Returns:

    Type Description dict[str, Any]

    dict[str, Any]: Dictionary containing extracted metadata.

    Notes

    Responsibilities:

    - Extract high-level metadata from the HTML document.\n- This includes: Document title, `<meta>` tag name/property to\n  content mappings.\n
    "},{"location":"omniread/html/parser/#omniread.html.parser.HTMLParser.parse_table","title":"parse_table staticmethod","text":"
    parse_table(table: Tag) -> list[list[str]]\n

    Parse an HTML table into a 2D list of strings.

    Parameters:

    Name Type Description Default table Tag

    BeautifulSoup tag representing a <table>.

    required

    Returns:

    Type Description list[list[str]]

    list[list[str]]: A list of rows, where each row is a list of cell text values.

    "},{"location":"omniread/html/parser/#omniread.html.parser.HTMLParser.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"omniread/html/scraper/","title":"Scraper","text":""},{"location":"omniread/html/scraper/#omniread.html.scraper","title":"omniread.html.scraper","text":""},{"location":"omniread/html/scraper/#omniread.html.scraper--summary","title":"Summary","text":"

    HTML scraping implementation for OmniRead.

    This module provides an HTTP-based scraper for retrieving HTML documents. It implements the core BaseScraper contract using httpx as the transport layer.

    This scraper is responsible for:

    • Fetching raw HTML bytes over HTTP(S)
    • Validating response content type
    • Attaching HTTP metadata to the returned content

    This scraper is not responsible for:

    • Parsing or interpreting HTML
    • Retrying failed requests
    • Managing crawl policies or rate limiting
    "},{"location":"omniread/html/scraper/#omniread.html.scraper-classes","title":"Classes","text":""},{"location":"omniread/html/scraper/#omniread.html.scraper.HTMLScraper","title":"HTMLScraper","text":"
    HTMLScraper(\n    *,\n    client: httpx.Client | None = None,\n    timeout: float = 15.0,\n    headers: Mapping[str, str] | None = None,\n    follow_redirects: bool = True\n)\n

    Bases: BaseScraper

    Base HTML scraper using httpx.

    Notes

    Responsibilities:

    - This scraper retrieves HTML documents over HTTP(S) and returns\n  them as raw content wrapped in a `Content` object.\n- Fetches raw bytes and metadata only.\n- The scraper uses `httpx.Client` for HTTP requests, enforces an\n  HTML content type, and preserves HTTP response metadata.\n

    Constraints:

    - The scraper does not: Parse HTML, perform retries or backoff,\n  handle non-HTML responses.\n

    Initialize the HTML scraper.

    Parameters:

    Name Type Description Default client Client | None

    Optional pre-configured httpx.Client. If omitted, a client is created internally.

    None timeout float

    Request timeout in seconds.

    15.0 headers Mapping[str, str] | None

    Optional default HTTP headers.

    None follow_redirects bool

    Whether to follow HTTP redirects.

    True"},{"location":"omniread/html/scraper/#omniread.html.scraper.HTMLScraper-functions","title":"Functions","text":""},{"location":"omniread/html/scraper/#omniread.html.scraper.HTMLScraper.fetch","title":"fetch","text":"
    fetch(\n    source: str,\n    *,\n    metadata: Mapping[str, Any] | None = None\n) -> Content\n

    Fetch an HTML document from the given source.

    Parameters:

    Name Type Description Default source str

    URL of the HTML document.

    required metadata Mapping[str, Any] | None

    Optional metadata to be merged into the returned content.

    None

    Returns:

    Name Type Description Content Content

    A Content instance containing raw HTML bytes, source URL, HTML content type, and HTTP response metadata.

    Raises:

    Type Description HTTPError

    If the HTTP request fails.

    ValueError

    If the response is not valid HTML.

    "},{"location":"omniread/html/scraper/#omniread.html.scraper.HTMLScraper.validate_content_type","title":"validate_content_type","text":"
    validate_content_type(response: httpx.Response) -> None\n

    Validate that the HTTP response contains HTML content.

    Parameters:

    Name Type Description Default response Response

    HTTP response returned by httpx.

    required

    Raises:

    Type Description ValueError

    If the Content-Type header is missing or does not indicate HTML content.

    "},{"location":"omniread/pdf/","title":"Pdf","text":"
    • Client
    • Parser
    • Scraper
    "},{"location":"omniread/pdf/#omniread.pdf","title":"omniread.pdf","text":""},{"location":"omniread/pdf/#omniread.pdf--summary","title":"Summary","text":"

    PDF format implementation for OmniRead.

    This package provides PDF-specific implementations of the core OmniRead contracts defined in omniread.core.

    Unlike HTML, PDF handling requires an explicit client layer for document access. This package therefore includes:

    • PDF clients for acquiring raw PDF data.
    • PDF scrapers that coordinate client access.
    • PDF parsers that extract structured content from PDF binaries.

    Public exports from this package represent the supported PDF pipeline and are safe for consumers to import directly when working with PDFs.

    "},{"location":"omniread/pdf/#omniread.pdf--public-api","title":"Public API","text":"
    • FileSystemPDFClient
    • PDFScraper
    • PDFParser
    "},{"location":"omniread/pdf/#omniread.pdf-classes","title":"Classes","text":""},{"location":"omniread/pdf/#omniread.pdf.FileSystemPDFClient","title":"FileSystemPDFClient","text":"

    Bases: BasePDFClient

    PDF client that reads from the local filesystem.

    Notes

    Guarantees:

    - This client reads PDF files directly from the disk and returns\n  their raw binary contents.\n
    "},{"location":"omniread/pdf/#omniread.pdf.FileSystemPDFClient-functions","title":"Functions","text":""},{"location":"omniread/pdf/#omniread.pdf.FileSystemPDFClient.fetch","title":"fetch","text":"
    fetch(path: Path) -> bytes\n

    Read a PDF file from the local filesystem.

    Parameters:

    Name Type Description Default path Path

    Filesystem path to the PDF file.

    required

    Returns:

    Name Type Description bytes bytes

    Raw PDF bytes.

    Raises:

    Type Description FileNotFoundError

    If the path does not exist.

    ValueError

    If the path exists but is not a file.

    "},{"location":"omniread/pdf/#omniread.pdf.PDFParser","title":"PDFParser","text":"
    PDFParser(content: Content)\n

    Bases: BaseParser[T], Generic[T]

    Base PDF parser.

    Notes

    Responsibilities:

    - This class enforces PDF content-type compatibility and provides\n  the extension point for implementing concrete PDF parsing strategies.\n

    Constraints:

    - Concrete implementations must define the output type `T` and\n  implement the `parse()` method.\n

    Initialize the parser with content to be parsed.

    Parameters:

    Name Type Description Default content Content

    Content instance to be parsed.

    required

    Raises:

    Type Description ValueError

    If the content type is not supported by this parser.

    "},{"location":"omniread/pdf/#omniread.pdf.PDFParser-attributes","title":"Attributes","text":""},{"location":"omniread/pdf/#omniread.pdf.PDFParser.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {PDF}\n

    Set of content types supported by this parser (PDF only).

    "},{"location":"omniread/pdf/#omniread.pdf.PDFParser-functions","title":"Functions","text":""},{"location":"omniread/pdf/#omniread.pdf.PDFParser.parse","title":"parse abstractmethod","text":"
    parse() -> T\n

    Parse PDF content into a structured output.

    Returns:

    Name Type Description T T

    Parsed representation of type T.

    Raises:

    Type Description Exception

    Parsing-specific errors as defined by the implementation.

    Notes

    Responsibilities:

    - Implementations must fully interpret the PDF binary payload and\n  return a deterministic, structured output.\n
    "},{"location":"omniread/pdf/#omniread.pdf.PDFParser.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"omniread/pdf/#omniread.pdf.PDFScraper","title":"PDFScraper","text":"
    PDFScraper(*, client: BasePDFClient)\n

    Bases: BaseScraper

    Scraper for PDF sources.

    Notes

    Responsibilities:

    - Delegates byte retrieval to a PDF client and normalizes output\n  into `Content`.\n- Preserves caller-provided metadata.\n

    Constraints:

    - The scraper does not perform parsing or interpretation.\n- Does not assume a specific storage backend.\n

    Initialize the PDF scraper.

    Parameters:

    Name Type Description Default client BasePDFClient

    PDF client responsible for retrieving raw PDF bytes.

    required"},{"location":"omniread/pdf/#omniread.pdf.PDFScraper-functions","title":"Functions","text":""},{"location":"omniread/pdf/#omniread.pdf.PDFScraper.fetch","title":"fetch","text":"
    fetch(\n    source: Any,\n    *,\n    metadata: Mapping[str, Any] | None = None\n) -> Content\n

    Fetch a PDF document from the given source.

    Parameters:

    Name Type Description Default source Any

    Identifier of the PDF source as understood by the configured PDF client.

    required metadata Mapping[str, Any] | None

    Optional metadata to attach to the returned content.

    None

    Returns:

    Name Type Description Content Content

    A Content instance containing raw PDF bytes, source identifier, PDF content type, and optional metadata.

    Raises:

    Type Description Exception

    Retrieval-specific errors raised by the PDF client.

    "},{"location":"omniread/pdf/client/","title":"Client","text":""},{"location":"omniread/pdf/client/#omniread.pdf.client","title":"omniread.pdf.client","text":""},{"location":"omniread/pdf/client/#omniread.pdf.client--summary","title":"Summary","text":"

    PDF client abstractions for OmniRead.

    This module defines the client layer responsible for retrieving raw PDF bytes from a concrete backing store.

    Clients provide low-level access to PDF binaries and are intentionally decoupled from scraping and parsing logic. They do not perform validation, interpretation, or content extraction.

    Typical backing stores include:

    • Local filesystems
    • Object storage (S3, GCS, etc.)
    • Network file systems
    "},{"location":"omniread/pdf/client/#omniread.pdf.client-classes","title":"Classes","text":""},{"location":"omniread/pdf/client/#omniread.pdf.client.BasePDFClient","title":"BasePDFClient","text":"

    Bases: ABC

    Abstract client responsible for retrieving PDF bytes.

    Retrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).

    Notes

    Responsibilities:

    - Implementations must accept a source identifier appropriate to\n  the backing store.\n- Return the full PDF binary payload.\n- Raise retrieval-specific errors on failure.\n
    "},{"location":"omniread/pdf/client/#omniread.pdf.client.BasePDFClient-functions","title":"Functions","text":""},{"location":"omniread/pdf/client/#omniread.pdf.client.BasePDFClient.fetch","title":"fetch abstractmethod","text":"
    fetch(source: Any) -> bytes\n

    Fetch raw PDF bytes from the given source.

    Parameters:

    Name Type Description Default source Any

    Identifier of the PDF location, such as a file path, object storage key, or remote reference.

    required

    Returns:

    Name Type Description bytes bytes

    Raw PDF bytes.

    Raises:

    Type Description Exception

    Retrieval-specific errors defined by the implementation.

    "},{"location":"omniread/pdf/client/#omniread.pdf.client.FileSystemPDFClient","title":"FileSystemPDFClient","text":"

    Bases: BasePDFClient

    PDF client that reads from the local filesystem.

    Notes

    Guarantees:

    - This client reads PDF files directly from the disk and returns\n  their raw binary contents.\n
    "},{"location":"omniread/pdf/client/#omniread.pdf.client.FileSystemPDFClient-functions","title":"Functions","text":""},{"location":"omniread/pdf/client/#omniread.pdf.client.FileSystemPDFClient.fetch","title":"fetch","text":"
    fetch(path: Path) -> bytes\n

    Read a PDF file from the local filesystem.

    Parameters:

    Name Type Description Default path Path

    Filesystem path to the PDF file.

    required

    Returns:

    Name Type Description bytes bytes

    Raw PDF bytes.

    Raises:

    Type Description FileNotFoundError

    If the path does not exist.

    ValueError

    If the path exists but is not a file.

    "},{"location":"omniread/pdf/parser/","title":"Parser","text":""},{"location":"omniread/pdf/parser/#omniread.pdf.parser","title":"omniread.pdf.parser","text":""},{"location":"omniread/pdf/parser/#omniread.pdf.parser--summary","title":"Summary","text":"

    PDF parser base implementations for OmniRead.

    This module defines the PDF-specific parser contract, extending the format-agnostic BaseParser with constraints appropriate for PDF content.

    PDF parsers are responsible for interpreting binary PDF data and producing structured representations suitable for downstream consumption.

    "},{"location":"omniread/pdf/parser/#omniread.pdf.parser-classes","title":"Classes","text":""},{"location":"omniread/pdf/parser/#omniread.pdf.parser.PDFParser","title":"PDFParser","text":"
    PDFParser(content: Content)\n

    Bases: BaseParser[T], Generic[T]

    Base PDF parser.

    Notes

    Responsibilities:

    - This class enforces PDF content-type compatibility and provides\n  the extension point for implementing concrete PDF parsing strategies.\n

    Constraints:

    - Concrete implementations must define the output type `T` and\n  implement the `parse()` method.\n

    Initialize the parser with content to be parsed.

    Parameters:

    Name Type Description Default content Content

    Content instance to be parsed.

    required

    Raises:

    Type Description ValueError

    If the content type is not supported by this parser.

    "},{"location":"omniread/pdf/parser/#omniread.pdf.parser.PDFParser-attributes","title":"Attributes","text":""},{"location":"omniread/pdf/parser/#omniread.pdf.parser.PDFParser.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {PDF}\n

    Set of content types supported by this parser (PDF only).

    "},{"location":"omniread/pdf/parser/#omniread.pdf.parser.PDFParser-functions","title":"Functions","text":""},{"location":"omniread/pdf/parser/#omniread.pdf.parser.PDFParser.parse","title":"parse abstractmethod","text":"
    parse() -> T\n

    Parse PDF content into a structured output.

    Returns:

    Name Type Description T T

    Parsed representation of type T.

    Raises:

    Type Description Exception

    Parsing-specific errors as defined by the implementation.

    Notes

    Responsibilities:

    - Implementations must fully interpret the PDF binary payload and\n  return a deterministic, structured output.\n
    "},{"location":"omniread/pdf/parser/#omniread.pdf.parser.PDFParser.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"omniread/pdf/scraper/","title":"Scraper","text":""},{"location":"omniread/pdf/scraper/#omniread.pdf.scraper","title":"omniread.pdf.scraper","text":""},{"location":"omniread/pdf/scraper/#omniread.pdf.scraper--summary","title":"Summary","text":"

    PDF scraping implementation for OmniRead.

    This module provides a PDF-specific scraper that coordinates PDF byte retrieval via a client and normalizes the result into a Content object.

    The scraper implements the core BaseScraper contract while delegating all storage and access concerns to a BasePDFClient implementation.

    "},{"location":"omniread/pdf/scraper/#omniread.pdf.scraper-classes","title":"Classes","text":""},{"location":"omniread/pdf/scraper/#omniread.pdf.scraper.PDFScraper","title":"PDFScraper","text":"
    PDFScraper(*, client: BasePDFClient)\n

    Bases: BaseScraper

    Scraper for PDF sources.

    Notes

    Responsibilities:

    - Delegates byte retrieval to a PDF client and normalizes output\n  into `Content`.\n- Preserves caller-provided metadata.\n

    Constraints:

    - The scraper does not perform parsing or interpretation.\n- Does not assume a specific storage backend.\n

    Initialize the PDF scraper.

    Parameters:

    Name Type Description Default client BasePDFClient

    PDF client responsible for retrieving raw PDF bytes.

    required"},{"location":"omniread/pdf/scraper/#omniread.pdf.scraper.PDFScraper-functions","title":"Functions","text":""},{"location":"omniread/pdf/scraper/#omniread.pdf.scraper.PDFScraper.fetch","title":"fetch","text":"
    fetch(\n    source: Any,\n    *,\n    metadata: Mapping[str, Any] | None = None\n) -> Content\n

    Fetch a PDF document from the given source.

    Parameters:

    Name Type Description Default source Any

    Identifier of the PDF source as understood by the configured PDF client.

    required metadata Mapping[str, Any] | None

    Optional metadata to attach to the returned content.

    None

    Returns:

    Name Type Description Content Content

    A Content instance containing raw PDF bytes, source identifier, PDF content type, and optional metadata.

    Raises:

    Type Description Exception

    Retrieval-specific errors raised by the PDF client.

    "},{"location":"omniread/xlsx/","title":"Xlsx","text":"
    • Client
    • Parser
    • Parser Base
    • Scraper
    "},{"location":"omniread/xlsx/#omniread.xlsx","title":"omniread.xlsx","text":""},{"location":"omniread/xlsx/#omniread.xlsx--summary","title":"Summary","text":"

    XLSX subpackage for OmniRead.

    Provides acquisition and parsing of Office Open XML spreadsheet (xlsx) content:

    • BaseXlsxClient: abstract backing-store client for xlsx bytes.
    • FileSystemXlsxClient: local filesystem implementation.
    • XlsxScraper: wraps fetched bytes into canonical Content.
    • XlsxParserBase: content-type-enforcing parser contract.
    • XlsxParser: generic string-row parser built on openpyxl.
    "},{"location":"omniread/xlsx/#omniread.xlsx-classes","title":"Classes","text":""},{"location":"omniread/xlsx/#omniread.xlsx.BaseXlsxClient","title":"BaseXlsxClient","text":"

    Bases: ABC

    Abstract client responsible for retrieving spreadsheet bytes.

    Retrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).

    Notes

    Responsibilities:

    - Implementations must accept a source identifier appropriate to\n  the backing store.\n- Return the full xlsx binary payload.\n- Raise retrieval-specific errors on failure.\n
    "},{"location":"omniread/xlsx/#omniread.xlsx.BaseXlsxClient-functions","title":"Functions","text":""},{"location":"omniread/xlsx/#omniread.xlsx.BaseXlsxClient.fetch","title":"fetch abstractmethod","text":"
    fetch(source: Any) -> bytes\n

    Fetch raw xlsx bytes from the given source.

    Parameters:

    Name Type Description Default source Any

    Identifier of the spreadsheet location, such as a file path, object storage key, or remote reference.

    required

    Returns:

    Name Type Description bytes bytes

    Raw xlsx bytes.

    Raises:

    Type Description Exception

    Retrieval-specific errors defined by the implementation.

    "},{"location":"omniread/xlsx/#omniread.xlsx.FileSystemXlsxClient","title":"FileSystemXlsxClient","text":"

    Bases: BaseXlsxClient

    XLSX client that reads from the local filesystem.

    Notes

    Guarantees:

    - This client reads spreadsheet files directly from the disk and\n  returns their raw binary contents.\n
    "},{"location":"omniread/xlsx/#omniread.xlsx.FileSystemXlsxClient-functions","title":"Functions","text":""},{"location":"omniread/xlsx/#omniread.xlsx.FileSystemXlsxClient.fetch","title":"fetch","text":"
    fetch(path: Path) -> bytes\n

    Read an xlsx file from the local filesystem.

    Parameters:

    Name Type Description Default path Path

    Filesystem path to the spreadsheet file.

    required

    Returns:

    Name Type Description bytes bytes

    Raw xlsx bytes.

    Raises:

    Type Description FileNotFoundError

    If the path does not exist.

    ValueError

    If the path exists but is not a file.

    "},{"location":"omniread/xlsx/#omniread.xlsx.XlsxParser","title":"XlsxParser","text":"
    XlsxParser(\n    content: Content,\n    *,\n    data_only: bool = True,\n    read_only: bool = True\n)\n

    Bases: XlsxParserBase[list[list[str]]]

    Generic xlsx parser producing string rows from a worksheet.

    Notes

    Responsibilities:

    - Lazily load the workbook owned by the parser's content.\n- Normalize cells (including dates and numeric values) into\n  deterministic string representations.\n- Expose sheet discovery and row extraction helpers.\n

    Constraints:

    - Cells are rendered with ``str(value)`` after trimming; date and\n  datetime values are rendered in ISO format. Consumers requiring\n  locale-specific formatting must convert on their side.\n

    Initialize the parser.

    Parameters:

    Name Type Description Default content Content

    XLSX content to parse; its type must be supported.

    required data_only bool

    Passed to openpyxl: when True, formula cells yield their last computed value instead of the formula string.

    True read_only bool

    Passed to openpyxl: streaming mode for lower memory usage.

    True"},{"location":"omniread/xlsx/#omniread.xlsx.XlsxParser-attributes","title":"Attributes","text":""},{"location":"omniread/xlsx/#omniread.xlsx.XlsxParser.sheet_names","title":"sheet_names property","text":"
    sheet_names: list[str]\n

    Names of all worksheets contained in the workbook.

    "},{"location":"omniread/xlsx/#omniread.xlsx.XlsxParser.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {XLSX}\n

    Set of content types supported by this parser (XLSX only).

    "},{"location":"omniread/xlsx/#omniread.xlsx.XlsxParser.workbook","title":"workbook property","text":"
    workbook: Workbook\n

    The lazily loaded workbook backing this parser's content.

    "},{"location":"omniread/xlsx/#omniread.xlsx.XlsxParser-functions","title":"Functions","text":""},{"location":"omniread/xlsx/#omniread.xlsx.XlsxParser.parse","title":"parse","text":"
    parse() -> list[list[str]]\n

    Parse the first worksheet into normalized string rows.

    Returns:

    Type Description list[list[str]]

    list[list[str]]: Rows of the default (first) worksheet.

    "},{"location":"omniread/xlsx/#omniread.xlsx.XlsxParser.rows","title":"rows","text":"
    rows(\n    sheet: int | str | None = None,\n    *,\n    skip_empty: bool = True\n) -> list[list[str]]\n

    Extract normalized string rows from a worksheet.

    Parameters:

    Name Type Description Default sheet int | str | None

    Worksheet index or title; defaults to the first worksheet.

    None skip_empty bool

    When True (default), rows whose cells are all blank are omitted.

    True

    Returns:

    Type Description list[list[str]]

    list[list[str]]: Normalized rows; trailing blank cells are trimmed per row.

    Raises:

    Type Description ValueError

    If the requested sheet does not exist.

    "},{"location":"omniread/xlsx/#omniread.xlsx.XlsxParser.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"omniread/xlsx/#omniread.xlsx.XlsxParserBase","title":"XlsxParserBase","text":"
    XlsxParserBase(content: Content)\n

    Bases: BaseParser[T], Generic[T]

    Base xlsx parser.

    Notes

    Responsibilities:

    - This class enforces xlsx content-type compatibility and provides\n  the extension point for implementing concrete xlsx parsing\n  strategies.\n

    Constraints:

    - Concrete implementations must define the output type `T` and\n  implement the `parse()` method.\n

    Initialize the parser with content to be parsed.

    Parameters:

    Name Type Description Default content Content

    Content instance to be parsed.

    required

    Raises:

    Type Description ValueError

    If the content type is not supported by this parser.

    "},{"location":"omniread/xlsx/#omniread.xlsx.XlsxParserBase-attributes","title":"Attributes","text":""},{"location":"omniread/xlsx/#omniread.xlsx.XlsxParserBase.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {XLSX}\n

    Set of content types supported by this parser (XLSX only).

    "},{"location":"omniread/xlsx/#omniread.xlsx.XlsxParserBase-functions","title":"Functions","text":""},{"location":"omniread/xlsx/#omniread.xlsx.XlsxParserBase.parse","title":"parse abstractmethod","text":"
    parse() -> T\n

    Parse xlsx content into a structured output.

    Returns:

    Name Type Description T T

    Parsed representation of type T.

    Raises:

    Type Description Exception

    Parsing-specific errors as defined by the implementation.

    "},{"location":"omniread/xlsx/#omniread.xlsx.XlsxParserBase.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"omniread/xlsx/#omniread.xlsx.XlsxScraper","title":"XlsxScraper","text":"
    XlsxScraper(*, client: BaseXlsxClient)\n

    Scraper for xlsx spreadsheet documents.

    Notes

    Responsibilities:

    - Fetch raw xlsx bytes via the configured client.\n- Wrap the payload in a canonical `Content` instance with the\n  XLSX content type and source identifier.\n

    Constraints:

    - The scraper does not perform parsing or interpretation.\n- Does not assume a specific storage backend.\n

    Initialize the XLSX scraper.

    Parameters:

    Name Type Description Default client BaseXlsxClient

    Client responsible for retrieving raw spreadsheet bytes.

    required"},{"location":"omniread/xlsx/#omniread.xlsx.XlsxScraper-functions","title":"Functions","text":""},{"location":"omniread/xlsx/#omniread.xlsx.XlsxScraper.fetch","title":"fetch","text":"
    fetch(\n    source: Any,\n    *,\n    metadata: Mapping[str, Any] | None = None\n) -> Content\n

    Fetch an xlsx document from the given source.

    Parameters:

    Name Type Description Default source Any

    Identifier of the spreadsheet source as understood by the configured client.

    required metadata Mapping[str, Any] | None

    Optional metadata to attach to the returned content.

    None

    Returns:

    Name Type Description Content Content

    A Content instance containing raw xlsx bytes, source identifier, XLSX content type, and optional metadata.

    Raises:

    Type Description Exception

    Retrieval-specific errors raised by the client.

    "},{"location":"omniread/xlsx/client/","title":"Client","text":""},{"location":"omniread/xlsx/client/#omniread.xlsx.client","title":"omniread.xlsx.client","text":""},{"location":"omniread/xlsx/client/#omniread.xlsx.client--summary","title":"Summary","text":"

    XLSX client abstractions for OmniRead.

    This module defines the client layer responsible for retrieving raw Office Open XML spreadsheet bytes from a concrete backing store.

    Clients provide low-level access to xlsx binaries and are intentionally decoupled from scraping and parsing logic. They do not perform validation, interpretation, or content extraction.

    Typical backing stores include:

    • Local filesystems
    • Object storage (S3, GCS, etc.)
    • Network file systems
    "},{"location":"omniread/xlsx/client/#omniread.xlsx.client-classes","title":"Classes","text":""},{"location":"omniread/xlsx/client/#omniread.xlsx.client.BaseXlsxClient","title":"BaseXlsxClient","text":"

    Bases: ABC

    Abstract client responsible for retrieving spreadsheet bytes.

    Retrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).

    Notes

    Responsibilities:

    - Implementations must accept a source identifier appropriate to\n  the backing store.\n- Return the full xlsx binary payload.\n- Raise retrieval-specific errors on failure.\n
    "},{"location":"omniread/xlsx/client/#omniread.xlsx.client.BaseXlsxClient-functions","title":"Functions","text":""},{"location":"omniread/xlsx/client/#omniread.xlsx.client.BaseXlsxClient.fetch","title":"fetch abstractmethod","text":"
    fetch(source: Any) -> bytes\n

    Fetch raw xlsx bytes from the given source.

    Parameters:

    Name Type Description Default source Any

    Identifier of the spreadsheet location, such as a file path, object storage key, or remote reference.

    required

    Returns:

    Name Type Description bytes bytes

    Raw xlsx bytes.

    Raises:

    Type Description Exception

    Retrieval-specific errors defined by the implementation.

    "},{"location":"omniread/xlsx/client/#omniread.xlsx.client.FileSystemXlsxClient","title":"FileSystemXlsxClient","text":"

    Bases: BaseXlsxClient

    XLSX client that reads from the local filesystem.

    Notes

    Guarantees:

    - This client reads spreadsheet files directly from the disk and\n  returns their raw binary contents.\n
    "},{"location":"omniread/xlsx/client/#omniread.xlsx.client.FileSystemXlsxClient-functions","title":"Functions","text":""},{"location":"omniread/xlsx/client/#omniread.xlsx.client.FileSystemXlsxClient.fetch","title":"fetch","text":"
    fetch(path: Path) -> bytes\n

    Read an xlsx file from the local filesystem.

    Parameters:

    Name Type Description Default path Path

    Filesystem path to the spreadsheet file.

    required

    Returns:

    Name Type Description bytes bytes

    Raw xlsx bytes.

    Raises:

    Type Description FileNotFoundError

    If the path does not exist.

    ValueError

    If the path exists but is not a file.

    "},{"location":"omniread/xlsx/parser/","title":"Parser","text":""},{"location":"omniread/xlsx/parser/#omniread.xlsx.parser","title":"omniread.xlsx.parser","text":""},{"location":"omniread/xlsx/parser/#omniread.xlsx.parser--summary","title":"Summary","text":"

    XLSX parser implementations for OmniRead.

    This module provides a concrete, generic parser for Office Open XML spreadsheets. It exposes workbook sheets as lists of string rows so downstream consumers can interpret tabular content without depending on openpyxl directly.

    The parser is intentionally statement-agnostic: it performs no header detection or column interpretation beyond basic cell normalization.

    "},{"location":"omniread/xlsx/parser/#omniread.xlsx.parser-classes","title":"Classes","text":""},{"location":"omniread/xlsx/parser/#omniread.xlsx.parser.XlsxParser","title":"XlsxParser","text":"
    XlsxParser(\n    content: Content,\n    *,\n    data_only: bool = True,\n    read_only: bool = True\n)\n

    Bases: XlsxParserBase[list[list[str]]]

    Generic xlsx parser producing string rows from a worksheet.

    Notes

    Responsibilities:

    - Lazily load the workbook owned by the parser's content.\n- Normalize cells (including dates and numeric values) into\n  deterministic string representations.\n- Expose sheet discovery and row extraction helpers.\n

    Constraints:

    - Cells are rendered with ``str(value)`` after trimming; date and\n  datetime values are rendered in ISO format. Consumers requiring\n  locale-specific formatting must convert on their side.\n

    Initialize the parser.

    Parameters:

    Name Type Description Default content Content

    XLSX content to parse; its type must be supported.

    required data_only bool

    Passed to openpyxl: when True, formula cells yield their last computed value instead of the formula string.

    True read_only bool

    Passed to openpyxl: streaming mode for lower memory usage.

    True"},{"location":"omniread/xlsx/parser/#omniread.xlsx.parser.XlsxParser-attributes","title":"Attributes","text":""},{"location":"omniread/xlsx/parser/#omniread.xlsx.parser.XlsxParser.sheet_names","title":"sheet_names property","text":"
    sheet_names: list[str]\n

    Names of all worksheets contained in the workbook.

    "},{"location":"omniread/xlsx/parser/#omniread.xlsx.parser.XlsxParser.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {XLSX}\n

    Set of content types supported by this parser (XLSX only).

    "},{"location":"omniread/xlsx/parser/#omniread.xlsx.parser.XlsxParser.workbook","title":"workbook property","text":"
    workbook: Workbook\n

    The lazily loaded workbook backing this parser's content.

    "},{"location":"omniread/xlsx/parser/#omniread.xlsx.parser.XlsxParser-functions","title":"Functions","text":""},{"location":"omniread/xlsx/parser/#omniread.xlsx.parser.XlsxParser.parse","title":"parse","text":"
    parse() -> list[list[str]]\n

    Parse the first worksheet into normalized string rows.

    Returns:

    Type Description list[list[str]]

    list[list[str]]: Rows of the default (first) worksheet.

    "},{"location":"omniread/xlsx/parser/#omniread.xlsx.parser.XlsxParser.rows","title":"rows","text":"
    rows(\n    sheet: int | str | None = None,\n    *,\n    skip_empty: bool = True\n) -> list[list[str]]\n

    Extract normalized string rows from a worksheet.

    Parameters:

    Name Type Description Default sheet int | str | None

    Worksheet index or title; defaults to the first worksheet.

    None skip_empty bool

    When True (default), rows whose cells are all blank are omitted.

    True

    Returns:

    Type Description list[list[str]]

    list[list[str]]: Normalized rows; trailing blank cells are trimmed per row.

    Raises:

    Type Description ValueError

    If the requested sheet does not exist.

    "},{"location":"omniread/xlsx/parser/#omniread.xlsx.parser.XlsxParser.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"omniread/xlsx/parser_base/","title":"Parser Base","text":""},{"location":"omniread/xlsx/parser_base/#omniread.xlsx.parser_base","title":"omniread.xlsx.parser_base","text":""},{"location":"omniread/xlsx/parser_base/#omniread.xlsx.parser_base--summary","title":"Summary","text":"

    XLSX parser base implementation for OmniRead.

    This module defines the XLSX-specific parser contract, extending the format-agnostic BaseParser with constraints appropriate for Office Open XML spreadsheet content.

    "},{"location":"omniread/xlsx/parser_base/#omniread.xlsx.parser_base-classes","title":"Classes","text":""},{"location":"omniread/xlsx/parser_base/#omniread.xlsx.parser_base.XlsxParserBase","title":"XlsxParserBase","text":"
    XlsxParserBase(content: Content)\n

    Bases: BaseParser[T], Generic[T]

    Base xlsx parser.

    Notes

    Responsibilities:

    - This class enforces xlsx content-type compatibility and provides\n  the extension point for implementing concrete xlsx parsing\n  strategies.\n

    Constraints:

    - Concrete implementations must define the output type `T` and\n  implement the `parse()` method.\n

    Initialize the parser with content to be parsed.

    Parameters:

    Name Type Description Default content Content

    Content instance to be parsed.

    required

    Raises:

    Type Description ValueError

    If the content type is not supported by this parser.

    "},{"location":"omniread/xlsx/parser_base/#omniread.xlsx.parser_base.XlsxParserBase-attributes","title":"Attributes","text":""},{"location":"omniread/xlsx/parser_base/#omniread.xlsx.parser_base.XlsxParserBase.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {XLSX}\n

    Set of content types supported by this parser (XLSX only).

    "},{"location":"omniread/xlsx/parser_base/#omniread.xlsx.parser_base.XlsxParserBase-functions","title":"Functions","text":""},{"location":"omniread/xlsx/parser_base/#omniread.xlsx.parser_base.XlsxParserBase.parse","title":"parse abstractmethod","text":"
    parse() -> T\n

    Parse xlsx content into a structured output.

    Returns:

    Name Type Description T T

    Parsed representation of type T.

    Raises:

    Type Description Exception

    Parsing-specific errors as defined by the implementation.

    "},{"location":"omniread/xlsx/parser_base/#omniread.xlsx.parser_base.XlsxParserBase.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"omniread/xlsx/scraper/","title":"Scraper","text":""},{"location":"omniread/xlsx/scraper/#omniread.xlsx.scraper","title":"omniread.xlsx.scraper","text":""},{"location":"omniread/xlsx/scraper/#omniread.xlsx.scraper--summary","title":"Summary","text":"

    XLSX scraper for OmniRead.

    This module defines the scraper responsible for acquiring raw Office Open XML spreadsheet content from a backing store via a configured client.

    The scraper does not interpret or parse the acquired bytes; it wraps them in the canonical Content model.

    "},{"location":"omniread/xlsx/scraper/#omniread.xlsx.scraper-classes","title":"Classes","text":""},{"location":"omniread/xlsx/scraper/#omniread.xlsx.scraper.XlsxScraper","title":"XlsxScraper","text":"
    XlsxScraper(*, client: BaseXlsxClient)\n

    Scraper for xlsx spreadsheet documents.

    Notes

    Responsibilities:

    - Fetch raw xlsx bytes via the configured client.\n- Wrap the payload in a canonical `Content` instance with the\n  XLSX content type and source identifier.\n

    Constraints:

    - The scraper does not perform parsing or interpretation.\n- Does not assume a specific storage backend.\n

    Initialize the XLSX scraper.

    Parameters:

    Name Type Description Default client BaseXlsxClient

    Client responsible for retrieving raw spreadsheet bytes.

    required"},{"location":"omniread/xlsx/scraper/#omniread.xlsx.scraper.XlsxScraper-functions","title":"Functions","text":""},{"location":"omniread/xlsx/scraper/#omniread.xlsx.scraper.XlsxScraper.fetch","title":"fetch","text":"
    fetch(\n    source: Any,\n    *,\n    metadata: Mapping[str, Any] | None = None\n) -> Content\n

    Fetch an xlsx document from the given source.

    Parameters:

    Name Type Description Default source Any

    Identifier of the spreadsheet source as understood by the configured client.

    required metadata Mapping[str, Any] | None

    Optional metadata to attach to the returned content.

    None

    Returns:

    Name Type Description Content Content

    A Content instance containing raw xlsx bytes, source identifier, XLSX content type, and optional metadata.

    Raises:

    Type Description Exception

    Retrieval-specific errors raised by the client.

    "},{"location":"pdf/","title":"Pdf","text":""},{"location":"pdf/#omniread.pdf","title":"omniread.pdf","text":""},{"location":"pdf/#omniread.pdf--summary","title":"Summary","text":"

    PDF format implementation for OmniRead.

    This package provides PDF-specific implementations of the core OmniRead contracts defined in omniread.core.

    Unlike HTML, PDF handling requires an explicit client layer for document access. This package therefore includes:

    • PDF clients for acquiring raw PDF data.
    • PDF scrapers that coordinate client access.
    • PDF parsers that extract structured content from PDF binaries.

    Public exports from this package represent the supported PDF pipeline and are safe for consumers to import directly when working with PDFs.

    "},{"location":"pdf/#omniread.pdf--public-api","title":"Public API","text":"
    • FileSystemPDFClient
    • PDFScraper
    • PDFParser
    "},{"location":"pdf/#omniread.pdf-classes","title":"Classes","text":""},{"location":"pdf/#omniread.pdf.FileSystemPDFClient","title":"FileSystemPDFClient","text":"

    Bases: BasePDFClient

    PDF client that reads from the local filesystem.

    Notes

    Guarantees:

    - This client reads PDF files directly from the disk and returns\n  their raw binary contents.\n
    "},{"location":"pdf/#omniread.pdf.FileSystemPDFClient-functions","title":"Functions","text":""},{"location":"pdf/#omniread.pdf.FileSystemPDFClient.fetch","title":"fetch","text":"
    fetch(path: Path) -> bytes\n

    Read a PDF file from the local filesystem.

    Parameters:

    Name Type Description Default path Path

    Filesystem path to the PDF file.

    required

    Returns:

    Name Type Description bytes bytes

    Raw PDF bytes.

    Raises:

    Type Description FileNotFoundError

    If the path does not exist.

    ValueError

    If the path exists but is not a file.

    "},{"location":"pdf/#omniread.pdf.PDFParser","title":"PDFParser","text":"
    PDFParser(content: Content)\n

    Bases: BaseParser[T], Generic[T]

    Base PDF parser.

    Notes

    Responsibilities:

    - This class enforces PDF content-type compatibility and provides\n  the extension point for implementing concrete PDF parsing strategies.\n

    Constraints:

    - Concrete implementations must define the output type `T` and\n  implement the `parse()` method.\n

    Initialize the parser with content to be parsed.

    Parameters:

    Name Type Description Default content Content

    Content instance to be parsed.

    required

    Raises:

    Type Description ValueError

    If the content type is not supported by this parser.

    "},{"location":"pdf/#omniread.pdf.PDFParser-attributes","title":"Attributes","text":""},{"location":"pdf/#omniread.pdf.PDFParser.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {PDF}\n

    Set of content types supported by this parser (PDF only).

    "},{"location":"pdf/#omniread.pdf.PDFParser-functions","title":"Functions","text":""},{"location":"pdf/#omniread.pdf.PDFParser.parse","title":"parse abstractmethod","text":"
    parse() -> T\n

    Parse PDF content into a structured output.

    Returns:

    Name Type Description T T

    Parsed representation of type T.

    Raises:

    Type Description Exception

    Parsing-specific errors as defined by the implementation.

    Notes

    Responsibilities:

    - Implementations must fully interpret the PDF binary payload and\n  return a deterministic, structured output.\n
    "},{"location":"pdf/#omniread.pdf.PDFParser.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"pdf/#omniread.pdf.PDFScraper","title":"PDFScraper","text":"
    PDFScraper(*, client: BasePDFClient)\n

    Bases: BaseScraper

    Scraper for PDF sources.

    Notes

    Responsibilities:

    - Delegates byte retrieval to a PDF client and normalizes output\n  into `Content`.\n- Preserves caller-provided metadata.\n

    Constraints:

    - The scraper does not perform parsing or interpretation.\n- Does not assume a specific storage backend.\n

    Initialize the PDF scraper.

    Parameters:

    Name Type Description Default client BasePDFClient

    PDF client responsible for retrieving raw PDF bytes.

    required"},{"location":"pdf/#omniread.pdf.PDFScraper-functions","title":"Functions","text":""},{"location":"pdf/#omniread.pdf.PDFScraper.fetch","title":"fetch","text":"
    fetch(\n    source: Any,\n    *,\n    metadata: Mapping[str, Any] | None = None\n) -> Content\n

    Fetch a PDF document from the given source.

    Parameters:

    Name Type Description Default source Any

    Identifier of the PDF source as understood by the configured PDF client.

    required metadata Mapping[str, Any] | None

    Optional metadata to attach to the returned content.

    None

    Returns:

    Name Type Description Content Content

    A Content instance containing raw PDF bytes, source identifier, PDF content type, and optional metadata.

    Raises:

    Type Description Exception

    Retrieval-specific errors raised by the PDF client.

    "},{"location":"pdf/client/","title":"Client","text":""},{"location":"pdf/client/#omniread.pdf.client","title":"omniread.pdf.client","text":""},{"location":"pdf/client/#omniread.pdf.client--summary","title":"Summary","text":"

    PDF client abstractions for OmniRead.

    This module defines the client layer responsible for retrieving raw PDF bytes from a concrete backing store.

    Clients provide low-level access to PDF binaries and are intentionally decoupled from scraping and parsing logic. They do not perform validation, interpretation, or content extraction.

    Typical backing stores include:

    • Local filesystems
    • Object storage (S3, GCS, etc.)
    • Network file systems
    "},{"location":"pdf/client/#omniread.pdf.client-classes","title":"Classes","text":""},{"location":"pdf/client/#omniread.pdf.client.BasePDFClient","title":"BasePDFClient","text":"

    Bases: ABC

    Abstract client responsible for retrieving PDF bytes.

    Retrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).

    Notes

    Responsibilities:

    - Implementations must accept a source identifier appropriate to\n  the backing store.\n- Return the full PDF binary payload.\n- Raise retrieval-specific errors on failure.\n
    "},{"location":"pdf/client/#omniread.pdf.client.BasePDFClient-functions","title":"Functions","text":""},{"location":"pdf/client/#omniread.pdf.client.BasePDFClient.fetch","title":"fetch abstractmethod","text":"
    fetch(source: Any) -> bytes\n

    Fetch raw PDF bytes from the given source.

    Parameters:

    Name Type Description Default source Any

    Identifier of the PDF location, such as a file path, object storage key, or remote reference.

    required

    Returns:

    Name Type Description bytes bytes

    Raw PDF bytes.

    Raises:

    Type Description Exception

    Retrieval-specific errors defined by the implementation.

    "},{"location":"pdf/client/#omniread.pdf.client.FileSystemPDFClient","title":"FileSystemPDFClient","text":"

    Bases: BasePDFClient

    PDF client that reads from the local filesystem.

    Notes

    Guarantees:

    - This client reads PDF files directly from the disk and returns\n  their raw binary contents.\n
    "},{"location":"pdf/client/#omniread.pdf.client.FileSystemPDFClient-functions","title":"Functions","text":""},{"location":"pdf/client/#omniread.pdf.client.FileSystemPDFClient.fetch","title":"fetch","text":"
    fetch(path: Path) -> bytes\n

    Read a PDF file from the local filesystem.

    Parameters:

    Name Type Description Default path Path

    Filesystem path to the PDF file.

    required

    Returns:

    Name Type Description bytes bytes

    Raw PDF bytes.

    Raises:

    Type Description FileNotFoundError

    If the path does not exist.

    ValueError

    If the path exists but is not a file.

    "},{"location":"pdf/parser/","title":"Parser","text":""},{"location":"pdf/parser/#omniread.pdf.parser","title":"omniread.pdf.parser","text":""},{"location":"pdf/parser/#omniread.pdf.parser--summary","title":"Summary","text":"

    PDF parser base implementations for OmniRead.

    This module defines the PDF-specific parser contract, extending the format-agnostic BaseParser with constraints appropriate for PDF content.

    PDF parsers are responsible for interpreting binary PDF data and producing structured representations suitable for downstream consumption.

    "},{"location":"pdf/parser/#omniread.pdf.parser-classes","title":"Classes","text":""},{"location":"pdf/parser/#omniread.pdf.parser.PDFParser","title":"PDFParser","text":"
    PDFParser(content: Content)\n

    Bases: BaseParser[T], Generic[T]

    Base PDF parser.

    Notes

    Responsibilities:

    - This class enforces PDF content-type compatibility and provides\n  the extension point for implementing concrete PDF parsing strategies.\n

    Constraints:

    - Concrete implementations must define the output type `T` and\n  implement the `parse()` method.\n

    Initialize the parser with content to be parsed.

    Parameters:

    Name Type Description Default content Content

    Content instance to be parsed.

    required

    Raises:

    Type Description ValueError

    If the content type is not supported by this parser.

    "},{"location":"pdf/parser/#omniread.pdf.parser.PDFParser-attributes","title":"Attributes","text":""},{"location":"pdf/parser/#omniread.pdf.parser.PDFParser.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {PDF}\n

    Set of content types supported by this parser (PDF only).

    "},{"location":"pdf/parser/#omniread.pdf.parser.PDFParser-functions","title":"Functions","text":""},{"location":"pdf/parser/#omniread.pdf.parser.PDFParser.parse","title":"parse abstractmethod","text":"
    parse() -> T\n

    Parse PDF content into a structured output.

    Returns:

    Name Type Description T T

    Parsed representation of type T.

    Raises:

    Type Description Exception

    Parsing-specific errors as defined by the implementation.

    Notes

    Responsibilities:

    - Implementations must fully interpret the PDF binary payload and\n  return a deterministic, structured output.\n
    "},{"location":"pdf/parser/#omniread.pdf.parser.PDFParser.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"pdf/scraper/","title":"Scraper","text":""},{"location":"pdf/scraper/#omniread.pdf.scraper","title":"omniread.pdf.scraper","text":""},{"location":"pdf/scraper/#omniread.pdf.scraper--summary","title":"Summary","text":"

    PDF scraping implementation for OmniRead.

    This module provides a PDF-specific scraper that coordinates PDF byte retrieval via a client and normalizes the result into a Content object.

    The scraper implements the core BaseScraper contract while delegating all storage and access concerns to a BasePDFClient implementation.

    "},{"location":"pdf/scraper/#omniread.pdf.scraper-classes","title":"Classes","text":""},{"location":"pdf/scraper/#omniread.pdf.scraper.PDFScraper","title":"PDFScraper","text":"
    PDFScraper(*, client: BasePDFClient)\n

    Bases: BaseScraper

    Scraper for PDF sources.

    Notes

    Responsibilities:

    - Delegates byte retrieval to a PDF client and normalizes output\n  into `Content`.\n- Preserves caller-provided metadata.\n

    Constraints:

    - The scraper does not perform parsing or interpretation.\n- Does not assume a specific storage backend.\n

    Initialize the PDF scraper.

    Parameters:

    Name Type Description Default client BasePDFClient

    PDF client responsible for retrieving raw PDF bytes.

    required"},{"location":"pdf/scraper/#omniread.pdf.scraper.PDFScraper-functions","title":"Functions","text":""},{"location":"pdf/scraper/#omniread.pdf.scraper.PDFScraper.fetch","title":"fetch","text":"
    fetch(\n    source: Any,\n    *,\n    metadata: Mapping[str, Any] | None = None\n) -> Content\n

    Fetch a PDF document from the given source.

    Parameters:

    Name Type Description Default source Any

    Identifier of the PDF source as understood by the configured PDF client.

    required metadata Mapping[str, Any] | None

    Optional metadata to attach to the returned content.

    None

    Returns:

    Name Type Description Content Content

    A Content instance containing raw PDF bytes, source identifier, PDF content type, and optional metadata.

    Raises:

    Type Description Exception

    Retrieval-specific errors raised by the PDF client.

    "},{"location":"xlsx/","title":"Xlsx","text":""},{"location":"xlsx/#omniread.xlsx","title":"omniread.xlsx","text":""},{"location":"xlsx/#omniread.xlsx--summary","title":"Summary","text":"

    XLSX subpackage for OmniRead.

    Provides acquisition and parsing of Office Open XML spreadsheet (xlsx) content:

    • BaseXlsxClient: abstract backing-store client for xlsx bytes.
    • FileSystemXlsxClient: local filesystem implementation.
    • XlsxScraper: wraps fetched bytes into canonical Content.
    • XlsxParserBase: content-type-enforcing parser contract.
    • XlsxParser: generic string-row parser built on openpyxl.
    "},{"location":"xlsx/#omniread.xlsx-classes","title":"Classes","text":""},{"location":"xlsx/#omniread.xlsx.BaseXlsxClient","title":"BaseXlsxClient","text":"

    Bases: ABC

    Abstract client responsible for retrieving spreadsheet bytes.

    Retrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).

    Notes

    Responsibilities:

    - Implementations must accept a source identifier appropriate to\n  the backing store.\n- Return the full xlsx binary payload.\n- Raise retrieval-specific errors on failure.\n
    "},{"location":"xlsx/#omniread.xlsx.BaseXlsxClient-functions","title":"Functions","text":""},{"location":"xlsx/#omniread.xlsx.BaseXlsxClient.fetch","title":"fetch abstractmethod","text":"
    fetch(source: Any) -> bytes\n

    Fetch raw xlsx bytes from the given source.

    Parameters:

    Name Type Description Default source Any

    Identifier of the spreadsheet location, such as a file path, object storage key, or remote reference.

    required

    Returns:

    Name Type Description bytes bytes

    Raw xlsx bytes.

    Raises:

    Type Description Exception

    Retrieval-specific errors defined by the implementation.

    "},{"location":"xlsx/#omniread.xlsx.FileSystemXlsxClient","title":"FileSystemXlsxClient","text":"

    Bases: BaseXlsxClient

    XLSX client that reads from the local filesystem.

    Notes

    Guarantees:

    - This client reads spreadsheet files directly from the disk and\n  returns their raw binary contents.\n
    "},{"location":"xlsx/#omniread.xlsx.FileSystemXlsxClient-functions","title":"Functions","text":""},{"location":"xlsx/#omniread.xlsx.FileSystemXlsxClient.fetch","title":"fetch","text":"
    fetch(path: Path) -> bytes\n

    Read an xlsx file from the local filesystem.

    Parameters:

    Name Type Description Default path Path

    Filesystem path to the spreadsheet file.

    required

    Returns:

    Name Type Description bytes bytes

    Raw xlsx bytes.

    Raises:

    Type Description FileNotFoundError

    If the path does not exist.

    ValueError

    If the path exists but is not a file.

    "},{"location":"xlsx/#omniread.xlsx.XlsxParser","title":"XlsxParser","text":"
    XlsxParser(\n    content: Content,\n    *,\n    data_only: bool = True,\n    read_only: bool = True\n)\n

    Bases: XlsxParserBase[list[list[str]]]

    Generic xlsx parser producing string rows from a worksheet.

    Notes

    Responsibilities:

    - Lazily load the workbook owned by the parser's content.\n- Normalize cells (including dates and numeric values) into\n  deterministic string representations.\n- Expose sheet discovery and row extraction helpers.\n

    Constraints:

    - Cells are rendered with ``str(value)`` after trimming; date and\n  datetime values are rendered in ISO format. Consumers requiring\n  locale-specific formatting must convert on their side.\n

    Initialize the parser.

    Parameters:

    Name Type Description Default content Content

    XLSX content to parse; its type must be supported.

    required data_only bool

    Passed to openpyxl: when True, formula cells yield their last computed value instead of the formula string.

    True read_only bool

    Passed to openpyxl: streaming mode for lower memory usage.

    True"},{"location":"xlsx/#omniread.xlsx.XlsxParser-attributes","title":"Attributes","text":""},{"location":"xlsx/#omniread.xlsx.XlsxParser.sheet_names","title":"sheet_names property","text":"
    sheet_names: list[str]\n

    Names of all worksheets contained in the workbook.

    "},{"location":"xlsx/#omniread.xlsx.XlsxParser.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {XLSX}\n

    Set of content types supported by this parser (XLSX only).

    "},{"location":"xlsx/#omniread.xlsx.XlsxParser.workbook","title":"workbook property","text":"
    workbook: Workbook\n

    The lazily loaded workbook backing this parser's content.

    "},{"location":"xlsx/#omniread.xlsx.XlsxParser-functions","title":"Functions","text":""},{"location":"xlsx/#omniread.xlsx.XlsxParser.parse","title":"parse","text":"
    parse() -> list[list[str]]\n

    Parse the first worksheet into normalized string rows.

    Returns:

    Type Description list[list[str]]

    list[list[str]]: Rows of the default (first) worksheet.

    "},{"location":"xlsx/#omniread.xlsx.XlsxParser.rows","title":"rows","text":"
    rows(\n    sheet: int | str | None = None,\n    *,\n    skip_empty: bool = True\n) -> list[list[str]]\n

    Extract normalized string rows from a worksheet.

    Parameters:

    Name Type Description Default sheet int | str | None

    Worksheet index or title; defaults to the first worksheet.

    None skip_empty bool

    When True (default), rows whose cells are all blank are omitted.

    True

    Returns:

    Type Description list[list[str]]

    list[list[str]]: Normalized rows; trailing blank cells are trimmed per row.

    Raises:

    Type Description ValueError

    If the requested sheet does not exist.

    "},{"location":"xlsx/#omniread.xlsx.XlsxParser.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"xlsx/#omniread.xlsx.XlsxParserBase","title":"XlsxParserBase","text":"
    XlsxParserBase(content: Content)\n

    Bases: BaseParser[T], Generic[T]

    Base xlsx parser.

    Notes

    Responsibilities:

    - This class enforces xlsx content-type compatibility and provides\n  the extension point for implementing concrete xlsx parsing\n  strategies.\n

    Constraints:

    - Concrete implementations must define the output type `T` and\n  implement the `parse()` method.\n

    Initialize the parser with content to be parsed.

    Parameters:

    Name Type Description Default content Content

    Content instance to be parsed.

    required

    Raises:

    Type Description ValueError

    If the content type is not supported by this parser.

    "},{"location":"xlsx/#omniread.xlsx.XlsxParserBase-attributes","title":"Attributes","text":""},{"location":"xlsx/#omniread.xlsx.XlsxParserBase.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {XLSX}\n

    Set of content types supported by this parser (XLSX only).

    "},{"location":"xlsx/#omniread.xlsx.XlsxParserBase-functions","title":"Functions","text":""},{"location":"xlsx/#omniread.xlsx.XlsxParserBase.parse","title":"parse abstractmethod","text":"
    parse() -> T\n

    Parse xlsx content into a structured output.

    Returns:

    Name Type Description T T

    Parsed representation of type T.

    Raises:

    Type Description Exception

    Parsing-specific errors as defined by the implementation.

    "},{"location":"xlsx/#omniread.xlsx.XlsxParserBase.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"xlsx/#omniread.xlsx.XlsxScraper","title":"XlsxScraper","text":"
    XlsxScraper(*, client: BaseXlsxClient)\n

    Scraper for xlsx spreadsheet documents.

    Notes

    Responsibilities:

    - Fetch raw xlsx bytes via the configured client.\n- Wrap the payload in a canonical `Content` instance with the\n  XLSX content type and source identifier.\n

    Constraints:

    - The scraper does not perform parsing or interpretation.\n- Does not assume a specific storage backend.\n

    Initialize the XLSX scraper.

    Parameters:

    Name Type Description Default client BaseXlsxClient

    Client responsible for retrieving raw spreadsheet bytes.

    required"},{"location":"xlsx/#omniread.xlsx.XlsxScraper-functions","title":"Functions","text":""},{"location":"xlsx/#omniread.xlsx.XlsxScraper.fetch","title":"fetch","text":"
    fetch(\n    source: Any,\n    *,\n    metadata: Mapping[str, Any] | None = None\n) -> Content\n

    Fetch an xlsx document from the given source.

    Parameters:

    Name Type Description Default source Any

    Identifier of the spreadsheet source as understood by the configured client.

    required metadata Mapping[str, Any] | None

    Optional metadata to attach to the returned content.

    None

    Returns:

    Name Type Description Content Content

    A Content instance containing raw xlsx bytes, source identifier, XLSX content type, and optional metadata.

    Raises:

    Type Description Exception

    Retrieval-specific errors raised by the client.

    "},{"location":"xlsx/client/","title":"Client","text":""},{"location":"xlsx/client/#omniread.xlsx.client","title":"omniread.xlsx.client","text":""},{"location":"xlsx/client/#omniread.xlsx.client--summary","title":"Summary","text":"

    XLSX client abstractions for OmniRead.

    This module defines the client layer responsible for retrieving raw Office Open XML spreadsheet bytes from a concrete backing store.

    Clients provide low-level access to xlsx binaries and are intentionally decoupled from scraping and parsing logic. They do not perform validation, interpretation, or content extraction.

    Typical backing stores include:

    • Local filesystems
    • Object storage (S3, GCS, etc.)
    • Network file systems
    "},{"location":"xlsx/client/#omniread.xlsx.client-classes","title":"Classes","text":""},{"location":"xlsx/client/#omniread.xlsx.client.BaseXlsxClient","title":"BaseXlsxClient","text":"

    Bases: ABC

    Abstract client responsible for retrieving spreadsheet bytes.

    Retrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).

    Notes

    Responsibilities:

    - Implementations must accept a source identifier appropriate to\n  the backing store.\n- Return the full xlsx binary payload.\n- Raise retrieval-specific errors on failure.\n
    "},{"location":"xlsx/client/#omniread.xlsx.client.BaseXlsxClient-functions","title":"Functions","text":""},{"location":"xlsx/client/#omniread.xlsx.client.BaseXlsxClient.fetch","title":"fetch abstractmethod","text":"
    fetch(source: Any) -> bytes\n

    Fetch raw xlsx bytes from the given source.

    Parameters:

    Name Type Description Default source Any

    Identifier of the spreadsheet location, such as a file path, object storage key, or remote reference.

    required

    Returns:

    Name Type Description bytes bytes

    Raw xlsx bytes.

    Raises:

    Type Description Exception

    Retrieval-specific errors defined by the implementation.

    "},{"location":"xlsx/client/#omniread.xlsx.client.FileSystemXlsxClient","title":"FileSystemXlsxClient","text":"

    Bases: BaseXlsxClient

    XLSX client that reads from the local filesystem.

    Notes

    Guarantees:

    - This client reads spreadsheet files directly from the disk and\n  returns their raw binary contents.\n
    "},{"location":"xlsx/client/#omniread.xlsx.client.FileSystemXlsxClient-functions","title":"Functions","text":""},{"location":"xlsx/client/#omniread.xlsx.client.FileSystemXlsxClient.fetch","title":"fetch","text":"
    fetch(path: Path) -> bytes\n

    Read an xlsx file from the local filesystem.

    Parameters:

    Name Type Description Default path Path

    Filesystem path to the spreadsheet file.

    required

    Returns:

    Name Type Description bytes bytes

    Raw xlsx bytes.

    Raises:

    Type Description FileNotFoundError

    If the path does not exist.

    ValueError

    If the path exists but is not a file.

    "},{"location":"xlsx/parser/","title":"Parser","text":""},{"location":"xlsx/parser/#omniread.xlsx.parser","title":"omniread.xlsx.parser","text":""},{"location":"xlsx/parser/#omniread.xlsx.parser--summary","title":"Summary","text":"

    XLSX parser implementations for OmniRead.

    This module provides a concrete, generic parser for Office Open XML spreadsheets. It exposes workbook sheets as lists of string rows so downstream consumers can interpret tabular content without depending on openpyxl directly.

    The parser is intentionally statement-agnostic: it performs no header detection or column interpretation beyond basic cell normalization.

    "},{"location":"xlsx/parser/#omniread.xlsx.parser-classes","title":"Classes","text":""},{"location":"xlsx/parser/#omniread.xlsx.parser.XlsxParser","title":"XlsxParser","text":"
    XlsxParser(\n    content: Content,\n    *,\n    data_only: bool = True,\n    read_only: bool = True\n)\n

    Bases: XlsxParserBase[list[list[str]]]

    Generic xlsx parser producing string rows from a worksheet.

    Notes

    Responsibilities:

    - Lazily load the workbook owned by the parser's content.\n- Normalize cells (including dates and numeric values) into\n  deterministic string representations.\n- Expose sheet discovery and row extraction helpers.\n

    Constraints:

    - Cells are rendered with ``str(value)`` after trimming; date and\n  datetime values are rendered in ISO format. Consumers requiring\n  locale-specific formatting must convert on their side.\n

    Initialize the parser.

    Parameters:

    Name Type Description Default content Content

    XLSX content to parse; its type must be supported.

    required data_only bool

    Passed to openpyxl: when True, formula cells yield their last computed value instead of the formula string.

    True read_only bool

    Passed to openpyxl: streaming mode for lower memory usage.

    True"},{"location":"xlsx/parser/#omniread.xlsx.parser.XlsxParser-attributes","title":"Attributes","text":""},{"location":"xlsx/parser/#omniread.xlsx.parser.XlsxParser.sheet_names","title":"sheet_names property","text":"
    sheet_names: list[str]\n

    Names of all worksheets contained in the workbook.

    "},{"location":"xlsx/parser/#omniread.xlsx.parser.XlsxParser.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {XLSX}\n

    Set of content types supported by this parser (XLSX only).

    "},{"location":"xlsx/parser/#omniread.xlsx.parser.XlsxParser.workbook","title":"workbook property","text":"
    workbook: Workbook\n

    The lazily loaded workbook backing this parser's content.

    "},{"location":"xlsx/parser/#omniread.xlsx.parser.XlsxParser-functions","title":"Functions","text":""},{"location":"xlsx/parser/#omniread.xlsx.parser.XlsxParser.parse","title":"parse","text":"
    parse() -> list[list[str]]\n

    Parse the first worksheet into normalized string rows.

    Returns:

    Type Description list[list[str]]

    list[list[str]]: Rows of the default (first) worksheet.

    "},{"location":"xlsx/parser/#omniread.xlsx.parser.XlsxParser.rows","title":"rows","text":"
    rows(\n    sheet: int | str | None = None,\n    *,\n    skip_empty: bool = True\n) -> list[list[str]]\n

    Extract normalized string rows from a worksheet.

    Parameters:

    Name Type Description Default sheet int | str | None

    Worksheet index or title; defaults to the first worksheet.

    None skip_empty bool

    When True (default), rows whose cells are all blank are omitted.

    True

    Returns:

    Type Description list[list[str]]

    list[list[str]]: Normalized rows; trailing blank cells are trimmed per row.

    Raises:

    Type Description ValueError

    If the requested sheet does not exist.

    "},{"location":"xlsx/parser/#omniread.xlsx.parser.XlsxParser.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"xlsx/parser_base/","title":"Parser Base","text":""},{"location":"xlsx/parser_base/#omniread.xlsx.parser_base","title":"omniread.xlsx.parser_base","text":""},{"location":"xlsx/parser_base/#omniread.xlsx.parser_base--summary","title":"Summary","text":"

    XLSX parser base implementation for OmniRead.

    This module defines the XLSX-specific parser contract, extending the format-agnostic BaseParser with constraints appropriate for Office Open XML spreadsheet content.

    "},{"location":"xlsx/parser_base/#omniread.xlsx.parser_base-classes","title":"Classes","text":""},{"location":"xlsx/parser_base/#omniread.xlsx.parser_base.XlsxParserBase","title":"XlsxParserBase","text":"
    XlsxParserBase(content: Content)\n

    Bases: BaseParser[T], Generic[T]

    Base xlsx parser.

    Notes

    Responsibilities:

    - This class enforces xlsx content-type compatibility and provides\n  the extension point for implementing concrete xlsx parsing\n  strategies.\n

    Constraints:

    - Concrete implementations must define the output type `T` and\n  implement the `parse()` method.\n

    Initialize the parser with content to be parsed.

    Parameters:

    Name Type Description Default content Content

    Content instance to be parsed.

    required

    Raises:

    Type Description ValueError

    If the content type is not supported by this parser.

    "},{"location":"xlsx/parser_base/#omniread.xlsx.parser_base.XlsxParserBase-attributes","title":"Attributes","text":""},{"location":"xlsx/parser_base/#omniread.xlsx.parser_base.XlsxParserBase.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {XLSX}\n

    Set of content types supported by this parser (XLSX only).

    "},{"location":"xlsx/parser_base/#omniread.xlsx.parser_base.XlsxParserBase-functions","title":"Functions","text":""},{"location":"xlsx/parser_base/#omniread.xlsx.parser_base.XlsxParserBase.parse","title":"parse abstractmethod","text":"
    parse() -> T\n

    Parse xlsx content into a structured output.

    Returns:

    Name Type Description T T

    Parsed representation of type T.

    Raises:

    Type Description Exception

    Parsing-specific errors as defined by the implementation.

    "},{"location":"xlsx/parser_base/#omniread.xlsx.parser_base.XlsxParserBase.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"xlsx/scraper/","title":"Scraper","text":""},{"location":"xlsx/scraper/#omniread.xlsx.scraper","title":"omniread.xlsx.scraper","text":""},{"location":"xlsx/scraper/#omniread.xlsx.scraper--summary","title":"Summary","text":"

    XLSX scraper for OmniRead.

    This module defines the scraper responsible for acquiring raw Office Open XML spreadsheet content from a backing store via a configured client.

    The scraper does not interpret or parse the acquired bytes; it wraps them in the canonical Content model.

    "},{"location":"xlsx/scraper/#omniread.xlsx.scraper-classes","title":"Classes","text":""},{"location":"xlsx/scraper/#omniread.xlsx.scraper.XlsxScraper","title":"XlsxScraper","text":"
    XlsxScraper(*, client: BaseXlsxClient)\n

    Scraper for xlsx spreadsheet documents.

    Notes

    Responsibilities:

    - Fetch raw xlsx bytes via the configured client.\n- Wrap the payload in a canonical `Content` instance with the\n  XLSX content type and source identifier.\n

    Constraints:

    - The scraper does not perform parsing or interpretation.\n- Does not assume a specific storage backend.\n

    Initialize the XLSX scraper.

    Parameters:

    Name Type Description Default client BaseXlsxClient

    Client responsible for retrieving raw spreadsheet bytes.

    required"},{"location":"xlsx/scraper/#omniread.xlsx.scraper.XlsxScraper-functions","title":"Functions","text":""},{"location":"xlsx/scraper/#omniread.xlsx.scraper.XlsxScraper.fetch","title":"fetch","text":"
    fetch(\n    source: Any,\n    *,\n    metadata: Mapping[str, Any] | None = None\n) -> Content\n

    Fetch an xlsx document from the given source.

    Parameters:

    Name Type Description Default source Any

    Identifier of the spreadsheet source as understood by the configured client.

    required metadata Mapping[str, Any] | None

    Optional metadata to attach to the returned content.

    None

    Returns:

    Name Type Description Content Content

    A Content instance containing raw xlsx bytes, source identifier, XLSX content type, and optional metadata.

    Raises:

    Type Description Exception

    Retrieval-specific errors raised by the client.

    "}]} \ No newline at end of file +{"config":{"lang":["en"],"separator":"[\\s\\-]+","pipeline":["stopWordFilter"]},"docs":[{"location":"","title":"omniread","text":""},{"location":"#omniread","title":"omniread","text":""},{"location":"#omniread--summary","title":"Summary","text":"

    OmniRead \u2014 format-agnostic content acquisition and parsing framework.

    OmniRead provides a cleanly layered architecture for fetching, parsing, and normalizing content from heterogeneous sources such as HTML documents and PDF files.

    The library is structured around three core concepts:

    1. Content: A canonical, format-agnostic container representing raw content bytes and minimal contextual metadata.
    2. Scrapers: Components responsible for acquiring raw content from a source (HTTP, filesystem, object storage, etc.). Scrapers never interpret content.
    3. Parsers: Components responsible for interpreting acquired content and converting it into structured, typed representations.

    OmniRead deliberately separates these responsibilities to ensure:

    • Clear boundaries between IO and interpretation.
    • Replaceable implementations per format.
    • Predictable, testable behavior.
    "},{"location":"#omniread--installation","title":"Installation","text":"

    Install OmniRead using pip:

    pip install omniread\n

    Install OmniRead using Poetry:

    poetry add omniread\n

    "},{"location":"#omniread--quick-start","title":"Quick start","text":"Example

    HTML example:

    from omniread import HTMLScraper, HTMLParser\n\nscraper = HTMLScraper()\ncontent = scraper.fetch(\"https://example.com\")\n\nclass TitleParser(HTMLParser[str]):\n    def parse(self) -> str:\n        return self._soup.title.string\n\nparser = TitleParser(content)\ntitle = parser.parse()\n

    PDF example:

    from omniread import FileSystemPDFClient, PDFScraper, PDFParser\nfrom pathlib import Path\n\nclient = FileSystemPDFClient()\nscraper = PDFScraper(client=client)\ncontent = scraper.fetch(Path(\"document.pdf\"))\n\nclass TextPDFParser(PDFParser[str]):\n    def parse(self) -> str:\n        # implement PDF text extraction\n        ...\n\nparser = TextPDFParser(content)\nresult = parser.parse()\n

    "},{"location":"#omniread--public-api","title":"Public API","text":"

    This module re-exports the recommended public entry points of OmniRead. Consumers are encouraged to import from this namespace rather than from format-specific submodules directly, unless advanced customization is required.

    • Content: Canonical content model.
    • ContentType: Supported media types.
    • HTMLScraper: HTTP-based HTML acquisition.
    • HTMLParser: Base parser for HTML DOM interpretation.
    • FileSystemPDFClient: Local filesystem PDF access.
    • PDFScraper: PDF-specific content acquisition.
    • PDFParser: Base parser for PDF binary interpretation.
    • FileSystemXlsxClient: Local filesystem spreadsheet access.
    • XlsxScraper: XLSX-specific content acquisition.
    • XlsxParser: Generic string-row parser for xlsx workbooks.
    "},{"location":"#omniread--core-philosophy","title":"Core Philosophy","text":"

    OmniRead is designed as a decoupled content engine:

    1. Separation of Concerns: Scrapers fetch, Parsers interpret. Neither knows about the other.
    2. Normalized Exchange: All components communicate via the Content model, ensuring a consistent contract.
    3. Format Agnosticism: The core logic is independent of whether the input is HTML, PDF, or JSON.
    "},{"location":"#omniread-classes","title":"Classes","text":""},{"location":"#omniread.Content","title":"Content dataclass","text":"
    Content(\n    raw: bytes,\n    source: str,\n    content_type: ContentType | None = ...,\n    metadata: Mapping[str, Any] | None = ...,\n)\n

    Normalized representation of extracted content.

    Notes

    Responsibilities:

    - A `Content` instance represents a raw content payload along with\n  minimal contextual metadata describing its origin and type.\n- This class is the primary exchange format between scrapers,\n  parsers, and downstream consumers.\n
    "},{"location":"#omniread.Content-attributes","title":"Attributes","text":""},{"location":"#omniread.Content.content_type","title":"content_type class-attribute instance-attribute","text":"
    content_type: ContentType | None = None\n

    Optional MIME type of the content, if known.

    "},{"location":"#omniread.Content.metadata","title":"metadata class-attribute instance-attribute","text":"
    metadata: Mapping[str, Any] | None = None\n

    Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes).

    "},{"location":"#omniread.Content.raw","title":"raw instance-attribute","text":"
    raw: bytes\n

    Raw content bytes as retrieved from the source.

    "},{"location":"#omniread.Content.source","title":"source instance-attribute","text":"
    source: str\n

    Identifier of the content origin (URL, file path, or logical name).

    "},{"location":"#omniread.ContentType","title":"ContentType","text":"

    Bases: str, Enum

    Supported MIME types for extracted content.

    Notes

    Guarantees:

    - This enum represents the declared or inferred media type of the\n  content source.\n- It is primarily used for routing content to the appropriate\n  parser or downstream consumer.\n
    "},{"location":"#omniread.ContentType-attributes","title":"Attributes","text":""},{"location":"#omniread.ContentType.CSV","title":"CSV class-attribute instance-attribute","text":"
    CSV = 'text/csv'\n

    Comma-separated-value document content.

    "},{"location":"#omniread.ContentType.HTML","title":"HTML class-attribute instance-attribute","text":"
    HTML = 'text/html'\n

    HTML document content.

    "},{"location":"#omniread.ContentType.JSON","title":"JSON class-attribute instance-attribute","text":"
    JSON = 'application/json'\n

    JSON document content.

    "},{"location":"#omniread.ContentType.PDF","title":"PDF class-attribute instance-attribute","text":"
    PDF = 'application/pdf'\n

    PDF document content.

    "},{"location":"#omniread.ContentType.XLSX","title":"XLSX class-attribute instance-attribute","text":"
    XLSX = \"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet\"\n

    Office Open XML spreadsheet (xlsx/xlsm) content.

    "},{"location":"#omniread.ContentType.XML","title":"XML class-attribute instance-attribute","text":"
    XML = 'application/xml'\n

    XML document content.

    "},{"location":"#omniread.FileSystemPDFClient","title":"FileSystemPDFClient","text":"

    Bases: BasePDFClient

    PDF client that reads from the local filesystem.

    Notes

    Guarantees:

    - This client reads PDF files directly from the disk and returns\n  their raw binary contents.\n
    "},{"location":"#omniread.FileSystemPDFClient-functions","title":"Functions","text":""},{"location":"#omniread.FileSystemPDFClient.fetch","title":"fetch","text":"
    fetch(path: Path) -> bytes\n

    Read a PDF file from the local filesystem.

    Parameters:

    Name Type Description Default path Path

    Filesystem path to the PDF file.

    required

    Returns:

    Name Type Description bytes bytes

    Raw PDF bytes.

    Raises:

    Type Description FileNotFoundError

    If the path does not exist.

    ValueError

    If the path exists but is not a file.

    "},{"location":"#omniread.HTMLParser","title":"HTMLParser","text":"
    HTMLParser(content: Content, features: str = 'html.parser')\n

    Bases: BaseParser[T], Generic[T]

    Base HTML parser.

    Notes

    Responsibilities:

    - This class extends the core `BaseParser` with HTML-specific behavior,\n  including DOM parsing via BeautifulSoup and reusable extraction helpers.\n- Provides reusable helpers for HTML extraction. Concrete parsers must\n  explicitly define the return type.\n

    Guarantees:

    - Accepts only HTML content.\n- Owns a parsed BeautifulSoup DOM tree.\n- Provides pure helper utilities for common HTML structures.\n

    Constraints:

    - Concrete subclasses must define the output type `T` and implement\n  the `parse()` method.\n

    Initialize the HTML parser.

    Parameters:

    Name Type Description Default content Content

    HTML content to be parsed.

    required features str

    BeautifulSoup parser backend to use (e.g., 'html.parser', 'lxml').

    'html.parser'

    Raises:

    Type Description ValueError

    If the content is empty or not valid HTML.

    "},{"location":"#omniread.HTMLParser-attributes","title":"Attributes","text":""},{"location":"#omniread.HTMLParser.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {HTML}\n

    Set of content types supported by this parser (HTML only).

    "},{"location":"#omniread.HTMLParser-functions","title":"Functions","text":""},{"location":"#omniread.HTMLParser.parse","title":"parse abstractmethod","text":"
    parse() -> T\n

    Fully parse the HTML content into structured output.

    Returns:

    Name Type Description T T

    Parsed representation of type T.

    Notes

    Responsibilities:

    - Implementations must fully interpret the HTML DOM and return a\n  deterministic, structured output.\n
    "},{"location":"#omniread.HTMLParser.parse_div","title":"parse_div staticmethod","text":"
    parse_div(div: Tag, *, separator: str = ' ') -> str\n

    Extract normalized text from a <div> element.

    Parameters:

    Name Type Description Default div Tag

    BeautifulSoup tag representing a <div>.

    required separator str

    String used to separate text nodes.

    ' '

    Returns:

    Name Type Description str str

    Flattened, whitespace-normalized text content.

    "},{"location":"#omniread.HTMLParser.parse_link","title":"parse_link staticmethod","text":"
    parse_link(a: Tag) -> str | None\n

    Extract the hyperlink reference from an <a> element.

    Parameters:

    Name Type Description Default a Tag

    BeautifulSoup tag representing an anchor.

    required

    Returns:

    Type Description str | None

    str | None: The value of the href attribute, or None if absent.

    "},{"location":"#omniread.HTMLParser.parse_meta","title":"parse_meta","text":"
    parse_meta() -> dict[str, Any]\n

    Extract high-level metadata from the HTML document.

    Returns:

    Type Description dict[str, Any]

    dict[str, Any]: Dictionary containing extracted metadata.

    Notes

    Responsibilities:

    - Extract high-level metadata from the HTML document.\n- This includes: Document title, `<meta>` tag name/property to\n  content mappings.\n
    "},{"location":"#omniread.HTMLParser.parse_table","title":"parse_table staticmethod","text":"
    parse_table(table: Tag) -> list[list[str]]\n

    Parse an HTML table into a 2D list of strings.

    Parameters:

    Name Type Description Default table Tag

    BeautifulSoup tag representing a <table>.

    required

    Returns:

    Type Description list[list[str]]

    list[list[str]]: A list of rows, where each row is a list of cell text values.

    "},{"location":"#omniread.HTMLParser.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"#omniread.HTMLScraper","title":"HTMLScraper","text":"
    HTMLScraper(\n    *,\n    client: httpx.Client | None = None,\n    timeout: float = 15.0,\n    headers: Mapping[str, str] | None = None,\n    follow_redirects: bool = True\n)\n

    Bases: BaseScraper

    Base HTML scraper using httpx.

    Notes

    Responsibilities:

    - This scraper retrieves HTML documents over HTTP(S) and returns\n  them as raw content wrapped in a `Content` object.\n- Fetches raw bytes and metadata only.\n- The scraper uses `httpx.Client` for HTTP requests, enforces an\n  HTML content type, and preserves HTTP response metadata.\n

    Constraints:

    - The scraper does not: Parse HTML, perform retries or backoff,\n  handle non-HTML responses.\n

    Initialize the HTML scraper.

    Parameters:

    Name Type Description Default client Client | None

    Optional pre-configured httpx.Client. If omitted, a client is created internally.

    None timeout float

    Request timeout in seconds.

    15.0 headers Mapping[str, str] | None

    Optional default HTTP headers.

    None follow_redirects bool

    Whether to follow HTTP redirects.

    True"},{"location":"#omniread.HTMLScraper-functions","title":"Functions","text":""},{"location":"#omniread.HTMLScraper.fetch","title":"fetch","text":"
    fetch(\n    source: str,\n    *,\n    metadata: Mapping[str, Any] | None = None\n) -> Content\n

    Fetch an HTML document from the given source.

    Parameters:

    Name Type Description Default source str

    URL of the HTML document.

    required metadata Mapping[str, Any] | None

    Optional metadata to be merged into the returned content.

    None

    Returns:

    Name Type Description Content Content

    A Content instance containing raw HTML bytes, source URL, HTML content type, and HTTP response metadata.

    Raises:

    Type Description HTTPError

    If the HTTP request fails.

    ValueError

    If the response is not valid HTML.

    "},{"location":"#omniread.HTMLScraper.validate_content_type","title":"validate_content_type","text":"
    validate_content_type(response: httpx.Response) -> None\n

    Validate that the HTTP response contains HTML content.

    Parameters:

    Name Type Description Default response Response

    HTTP response returned by httpx.

    required

    Raises:

    Type Description ValueError

    If the Content-Type header is missing or does not indicate HTML content.

    "},{"location":"#omniread.PDFParser","title":"PDFParser","text":"
    PDFParser(content: Content)\n

    Bases: BaseParser[T], Generic[T]

    Base PDF parser.

    Notes

    Responsibilities:

    - This class enforces PDF content-type compatibility and provides\n  the extension point for implementing concrete PDF parsing strategies.\n

    Constraints:

    - Concrete implementations must define the output type `T` and\n  implement the `parse()` method.\n

    Initialize the parser with content to be parsed.

    Parameters:

    Name Type Description Default content Content

    Content instance to be parsed.

    required

    Raises:

    Type Description ValueError

    If the content type is not supported by this parser.

    "},{"location":"#omniread.PDFParser-attributes","title":"Attributes","text":""},{"location":"#omniread.PDFParser.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {PDF}\n

    Set of content types supported by this parser (PDF only).

    "},{"location":"#omniread.PDFParser-functions","title":"Functions","text":""},{"location":"#omniread.PDFParser.parse","title":"parse abstractmethod","text":"
    parse() -> T\n

    Parse PDF content into a structured output.

    Returns:

    Name Type Description T T

    Parsed representation of type T.

    Raises:

    Type Description Exception

    Parsing-specific errors as defined by the implementation.

    Notes

    Responsibilities:

    - Implementations must fully interpret the PDF binary payload and\n  return a deterministic, structured output.\n
    "},{"location":"#omniread.PDFParser.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"#omniread.PDFScraper","title":"PDFScraper","text":"
    PDFScraper(*, client: BasePDFClient)\n

    Bases: BaseScraper

    Scraper for PDF sources.

    Notes

    Responsibilities:

    - Delegates byte retrieval to a PDF client and normalizes output\n  into `Content`.\n- Preserves caller-provided metadata.\n

    Constraints:

    - The scraper does not perform parsing or interpretation.\n- Does not assume a specific storage backend.\n

    Initialize the PDF scraper.

    Parameters:

    Name Type Description Default client BasePDFClient

    PDF client responsible for retrieving raw PDF bytes.

    required"},{"location":"#omniread.PDFScraper-functions","title":"Functions","text":""},{"location":"#omniread.PDFScraper.fetch","title":"fetch","text":"
    fetch(\n    source: Any,\n    *,\n    metadata: Mapping[str, Any] | None = None\n) -> Content\n

    Fetch a PDF document from the given source.

    Parameters:

    Name Type Description Default source Any

    Identifier of the PDF source as understood by the configured PDF client.

    required metadata Mapping[str, Any] | None

    Optional metadata to attach to the returned content.

    None

    Returns:

    Name Type Description Content Content

    A Content instance containing raw PDF bytes, source identifier, PDF content type, and optional metadata.

    Raises:

    Type Description Exception

    Retrieval-specific errors raised by the PDF client.

    "},{"location":"core/","title":"Core","text":""},{"location":"core/#omniread.core","title":"omniread.core","text":""},{"location":"core/#omniread.core--summary","title":"Summary","text":"

    Core domain contracts for OmniRead.

    This package defines the format-agnostic domain layer of OmniRead. It exposes canonical content models and abstract interfaces that are implemented by format-specific modules (HTML, PDF, etc.).

    Public exports from this package are considered stable contracts and are safe for downstream consumers to depend on.

    Submodules:

    • content: Canonical content models and enums.
    • parser: Abstract parsing contracts.
    • scraper: Abstract scraping contracts.

    Format-specific behavior must not be introduced at this layer.

    "},{"location":"core/#omniread.core--public-api","title":"Public API","text":"
    • Content
    • ContentType
    "},{"location":"core/#omniread.core-classes","title":"Classes","text":""},{"location":"core/#omniread.core.BaseParser","title":"BaseParser","text":"
    BaseParser(content: Content)\n

    Bases: ABC, Generic[T]

    Base interface for all parsers.

    Notes

    Guarantees:

    - A parser is a self-contained object that owns the `Content` it is\n  responsible for interpreting.\n- Consumers may rely on early validation of content compatibility\n  and type-stable return values from `parse()`.\n

    Responsibilities:

    - Implementations must declare supported content types via `supported_types`.\n- Implementations must raise parsing-specific exceptions from `parse()`.\n- Implementations must remain deterministic for a given input.\n

    Initialize the parser with content to be parsed.

    Parameters:

    Name Type Description Default content Content

    Content instance to be parsed.

    required

    Raises:

    Type Description ValueError

    If the content type is not supported by this parser.

    "},{"location":"core/#omniread.core.BaseParser-attributes","title":"Attributes","text":""},{"location":"core/#omniread.core.BaseParser.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = set()\n

    Set of content types supported by this parser. An empty set indicates that the parser is content-type agnostic.

    "},{"location":"core/#omniread.core.BaseParser-functions","title":"Functions","text":""},{"location":"core/#omniread.core.BaseParser.parse","title":"parse abstractmethod","text":"
    parse() -> T\n

    Parse the owned content into structured output.

    Returns:

    Name Type Description T T

    Parsed, structured representation.

    Raises:

    Type Description Exception

    Parsing-specific errors as defined by the implementation.

    Notes

    Responsibilities:

    - Implementations must fully consume the provided content and\n  return a deterministic, structured output.\n
    "},{"location":"core/#omniread.core.BaseParser.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"core/#omniread.core.BaseScraper","title":"BaseScraper","text":"

    Bases: ABC

    Base interface for all scrapers.

    Notes

    Responsibilities:

    - A scraper is responsible ONLY for fetching raw content (bytes)\n  from a source. It must not interpret or parse it.\n- A scraper is a stateless acquisition component that retrieves raw\n  content from a source and returns it as a `Content` object.\n- Scrapers define how content is obtained, not what the content means.\n- Implementations may vary in transport mechanism, authentication\n  strategy, retry and backoff behavior.\n

    Constraints:

    - Implementations must not parse content, modify content semantics,\n  or couple scraping logic to a specific parser.\n
    "},{"location":"core/#omniread.core.BaseScraper-functions","title":"Functions","text":""},{"location":"core/#omniread.core.BaseScraper.fetch","title":"fetch abstractmethod","text":"
    fetch(\n    source: str,\n    *,\n    metadata: Mapping[str, Any] | None = None\n) -> Content\n

    Fetch raw content from the given source.

    Parameters:

    Name Type Description Default source str

    Location identifier (URL, file path, S3 URI, etc.).

    required metadata Mapping[str, Any] | None

    Optional hints for the scraper (headers, auth, etc.).

    None

    Returns:

    Name Type Description Content Content

    Content object containing raw bytes and metadata.

    Raises:

    Type Description Exception

    Retrieval-specific errors as defined by the implementation.

    Notes

    Responsibilities:

    - Implementations must retrieve the content referenced by `source`\n  and return it as raw bytes wrapped in a `Content` object.\n
    "},{"location":"core/#omniread.core.Content","title":"Content dataclass","text":"
    Content(\n    raw: bytes,\n    source: str,\n    content_type: ContentType | None = ...,\n    metadata: Mapping[str, Any] | None = ...,\n)\n

    Normalized representation of extracted content.

    Notes

    Responsibilities:

    - A `Content` instance represents a raw content payload along with\n  minimal contextual metadata describing its origin and type.\n- This class is the primary exchange format between scrapers,\n  parsers, and downstream consumers.\n
    "},{"location":"core/#omniread.core.Content-attributes","title":"Attributes","text":""},{"location":"core/#omniread.core.Content.content_type","title":"content_type class-attribute instance-attribute","text":"
    content_type: ContentType | None = None\n

    Optional MIME type of the content, if known.

    "},{"location":"core/#omniread.core.Content.metadata","title":"metadata class-attribute instance-attribute","text":"
    metadata: Mapping[str, Any] | None = None\n

    Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes).

    "},{"location":"core/#omniread.core.Content.raw","title":"raw instance-attribute","text":"
    raw: bytes\n

    Raw content bytes as retrieved from the source.

    "},{"location":"core/#omniread.core.Content.source","title":"source instance-attribute","text":"
    source: str\n

    Identifier of the content origin (URL, file path, or logical name).

    "},{"location":"core/#omniread.core.ContentType","title":"ContentType","text":"

    Bases: str, Enum

    Supported MIME types for extracted content.

    Notes

    Guarantees:

    - This enum represents the declared or inferred media type of the\n  content source.\n- It is primarily used for routing content to the appropriate\n  parser or downstream consumer.\n
    "},{"location":"core/#omniread.core.ContentType-attributes","title":"Attributes","text":""},{"location":"core/#omniread.core.ContentType.CSV","title":"CSV class-attribute instance-attribute","text":"
    CSV = 'text/csv'\n

    Comma-separated-value document content.

    "},{"location":"core/#omniread.core.ContentType.HTML","title":"HTML class-attribute instance-attribute","text":"
    HTML = 'text/html'\n

    HTML document content.

    "},{"location":"core/#omniread.core.ContentType.JSON","title":"JSON class-attribute instance-attribute","text":"
    JSON = 'application/json'\n

    JSON document content.

    "},{"location":"core/#omniread.core.ContentType.PDF","title":"PDF class-attribute instance-attribute","text":"
    PDF = 'application/pdf'\n

    PDF document content.

    "},{"location":"core/#omniread.core.ContentType.XLSX","title":"XLSX class-attribute instance-attribute","text":"
    XLSX = \"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet\"\n

    Office Open XML spreadsheet (xlsx/xlsm) content.

    "},{"location":"core/#omniread.core.ContentType.XML","title":"XML class-attribute instance-attribute","text":"
    XML = 'application/xml'\n

    XML document content.

    "},{"location":"core/content/","title":"Content","text":""},{"location":"core/content/#omniread.core.content","title":"omniread.core.content","text":""},{"location":"core/content/#omniread.core.content--summary","title":"Summary","text":"

    Canonical content models for OmniRead.

    This module defines the format-agnostic content representation used across all parsers and scrapers in OmniRead.

    The models defined here represent what was extracted, not how it was retrieved or parsed. Format-specific behavior and metadata must not alter the semantic meaning of these models.

    "},{"location":"core/content/#omniread.core.content-classes","title":"Classes","text":""},{"location":"core/content/#omniread.core.content.Content","title":"Content dataclass","text":"
    Content(\n    raw: bytes,\n    source: str,\n    content_type: ContentType | None = ...,\n    metadata: Mapping[str, Any] | None = ...,\n)\n

    Normalized representation of extracted content.

    Notes

    Responsibilities:

    - A `Content` instance represents a raw content payload along with\n  minimal contextual metadata describing its origin and type.\n- This class is the primary exchange format between scrapers,\n  parsers, and downstream consumers.\n
    "},{"location":"core/content/#omniread.core.content.Content-attributes","title":"Attributes","text":""},{"location":"core/content/#omniread.core.content.Content.content_type","title":"content_type class-attribute instance-attribute","text":"
    content_type: ContentType | None = None\n

    Optional MIME type of the content, if known.

    "},{"location":"core/content/#omniread.core.content.Content.metadata","title":"metadata class-attribute instance-attribute","text":"
    metadata: Mapping[str, Any] | None = None\n

    Optional, implementation-defined metadata associated with the content (e.g., headers, encoding hints, extraction notes).

    "},{"location":"core/content/#omniread.core.content.Content.raw","title":"raw instance-attribute","text":"
    raw: bytes\n

    Raw content bytes as retrieved from the source.

    "},{"location":"core/content/#omniread.core.content.Content.source","title":"source instance-attribute","text":"
    source: str\n

    Identifier of the content origin (URL, file path, or logical name).

    "},{"location":"core/content/#omniread.core.content.ContentType","title":"ContentType","text":"

    Bases: str, Enum

    Supported MIME types for extracted content.

    Notes

    Guarantees:

    - This enum represents the declared or inferred media type of the\n  content source.\n- It is primarily used for routing content to the appropriate\n  parser or downstream consumer.\n
    "},{"location":"core/content/#omniread.core.content.ContentType-attributes","title":"Attributes","text":""},{"location":"core/content/#omniread.core.content.ContentType.CSV","title":"CSV class-attribute instance-attribute","text":"
    CSV = 'text/csv'\n

    Comma-separated-value document content.

    "},{"location":"core/content/#omniread.core.content.ContentType.HTML","title":"HTML class-attribute instance-attribute","text":"
    HTML = 'text/html'\n

    HTML document content.

    "},{"location":"core/content/#omniread.core.content.ContentType.JSON","title":"JSON class-attribute instance-attribute","text":"
    JSON = 'application/json'\n

    JSON document content.

    "},{"location":"core/content/#omniread.core.content.ContentType.PDF","title":"PDF class-attribute instance-attribute","text":"
    PDF = 'application/pdf'\n

    PDF document content.

    "},{"location":"core/content/#omniread.core.content.ContentType.XLSX","title":"XLSX class-attribute instance-attribute","text":"
    XLSX = \"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet\"\n

    Office Open XML spreadsheet (xlsx/xlsm) content.

    "},{"location":"core/content/#omniread.core.content.ContentType.XML","title":"XML class-attribute instance-attribute","text":"
    XML = 'application/xml'\n

    XML document content.

    "},{"location":"core/parser/","title":"Parser","text":""},{"location":"core/parser/#omniread.core.parser","title":"omniread.core.parser","text":""},{"location":"core/parser/#omniread.core.parser--summary","title":"Summary","text":"

    Abstract parsing contracts for OmniRead.

    This module defines the format-agnostic parser interface used to transform raw content into structured, typed representations.

    Parsers are responsible for:

    • Interpreting a single Content instance
    • Validating compatibility with the content type
    • Producing a structured output suitable for downstream consumers

    Parsers are not responsible for:

    • Fetching or acquiring content
    • Performing retries or error recovery
    • Managing multiple content sources
    "},{"location":"core/parser/#omniread.core.parser-classes","title":"Classes","text":""},{"location":"core/parser/#omniread.core.parser.BaseParser","title":"BaseParser","text":"
    BaseParser(content: Content)\n

    Bases: ABC, Generic[T]

    Base interface for all parsers.

    Notes

    Guarantees:

    - A parser is a self-contained object that owns the `Content` it is\n  responsible for interpreting.\n- Consumers may rely on early validation of content compatibility\n  and type-stable return values from `parse()`.\n

    Responsibilities:

    - Implementations must declare supported content types via `supported_types`.\n- Implementations must raise parsing-specific exceptions from `parse()`.\n- Implementations must remain deterministic for a given input.\n

    Initialize the parser with content to be parsed.

    Parameters:

    Name Type Description Default content Content

    Content instance to be parsed.

    required

    Raises:

    Type Description ValueError

    If the content type is not supported by this parser.

    "},{"location":"core/parser/#omniread.core.parser.BaseParser-attributes","title":"Attributes","text":""},{"location":"core/parser/#omniread.core.parser.BaseParser.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = set()\n

    Set of content types supported by this parser. An empty set indicates that the parser is content-type agnostic.

    "},{"location":"core/parser/#omniread.core.parser.BaseParser-functions","title":"Functions","text":""},{"location":"core/parser/#omniread.core.parser.BaseParser.parse","title":"parse abstractmethod","text":"
    parse() -> T\n

    Parse the owned content into structured output.

    Returns:

    Name Type Description T T

    Parsed, structured representation.

    Raises:

    Type Description Exception

    Parsing-specific errors as defined by the implementation.

    Notes

    Responsibilities:

    - Implementations must fully consume the provided content and\n  return a deterministic, structured output.\n
    "},{"location":"core/parser/#omniread.core.parser.BaseParser.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"core/scraper/","title":"Scraper","text":""},{"location":"core/scraper/#omniread.core.scraper","title":"omniread.core.scraper","text":""},{"location":"core/scraper/#omniread.core.scraper--summary","title":"Summary","text":"

    Abstract scraping contracts for OmniRead.

    This module defines the format-agnostic scraper interface responsible for acquiring raw content from external sources.

    Scrapers are responsible for:

    • Locating and retrieving raw content bytes
    • Attaching minimal contextual metadata
    • Returning normalized Content objects

    Scrapers are explicitly NOT responsible for:

    • Parsing or interpreting content
    • Inferring structure or semantics
    • Performing content-type specific processing

    All interpretation must be delegated to parsers.

    "},{"location":"core/scraper/#omniread.core.scraper-classes","title":"Classes","text":""},{"location":"core/scraper/#omniread.core.scraper.BaseScraper","title":"BaseScraper","text":"

    Bases: ABC

    Base interface for all scrapers.

    Notes

    Responsibilities:

    - A scraper is responsible ONLY for fetching raw content (bytes)\n  from a source. It must not interpret or parse it.\n- A scraper is a stateless acquisition component that retrieves raw\n  content from a source and returns it as a `Content` object.\n- Scrapers define how content is obtained, not what the content means.\n- Implementations may vary in transport mechanism, authentication\n  strategy, retry and backoff behavior.\n

    Constraints:

    - Implementations must not parse content, modify content semantics,\n  or couple scraping logic to a specific parser.\n
    "},{"location":"core/scraper/#omniread.core.scraper.BaseScraper-functions","title":"Functions","text":""},{"location":"core/scraper/#omniread.core.scraper.BaseScraper.fetch","title":"fetch abstractmethod","text":"
    fetch(\n    source: str,\n    *,\n    metadata: Mapping[str, Any] | None = None\n) -> Content\n

    Fetch raw content from the given source.

    Parameters:

    Name Type Description Default source str

    Location identifier (URL, file path, S3 URI, etc.).

    required metadata Mapping[str, Any] | None

    Optional hints for the scraper (headers, auth, etc.).

    None

    Returns:

    Name Type Description Content Content

    Content object containing raw bytes and metadata.

    Raises:

    Type Description Exception

    Retrieval-specific errors as defined by the implementation.

    Notes

    Responsibilities:

    - Implementations must retrieve the content referenced by `source`\n  and return it as raw bytes wrapped in a `Content` object.\n
    "},{"location":"csv/","title":"Csv","text":""},{"location":"csv/#omniread.csv","title":"omniread.csv","text":""},{"location":"csv/#omniread.csv--summary","title":"Summary","text":"

    CSV subpackage for OmniRead.

    Provides acquisition and parsing of comma-separated-value content:

    • BaseCsvClient: abstract backing-store client for csv bytes.
    • FileSystemCsvClient: local filesystem implementation.
    • CsvScraper: wraps fetched bytes into canonical Content.
    • CsvParserBase: content-type-enforcing parser contract.
    • CsvParser: generic string-row parser built on the standard csv module.
    "},{"location":"csv/#omniread.csv-classes","title":"Classes","text":""},{"location":"csv/#omniread.csv.BaseCsvClient","title":"BaseCsvClient","text":"

    Bases: ABC

    Abstract client responsible for retrieving csv bytes.

    Retrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).

    Notes

    Responsibilities:

    - Implementations must accept a source identifier appropriate to\n  the backing store.\n- Return the full csv binary payload.\n- Raise retrieval-specific errors on failure.\n
    "},{"location":"csv/#omniread.csv.BaseCsvClient-functions","title":"Functions","text":""},{"location":"csv/#omniread.csv.BaseCsvClient.fetch","title":"fetch abstractmethod","text":"
    fetch(source: Any) -> bytes\n

    Fetch raw csv bytes from the given source.

    Parameters:

    Name Type Description Default source Any

    Identifier of the csv location, such as a file path, object storage key, or remote reference.

    required

    Returns:

    Name Type Description bytes bytes

    Raw csv bytes.

    Raises:

    Type Description Exception

    Retrieval-specific errors defined by the implementation.

    "},{"location":"csv/#omniread.csv.CsvParser","title":"CsvParser","text":"
    CsvParser(content: Content)\n

    Bases: CsvParserBase[list[list[str]]]

    Generic csv parser producing string rows from the document.

    Notes

    Responsibilities:

    - Decode the payload (UTF-8 with BOM support, Latin-1 fallback).\n- Detect the delimiter from a leading sample (`,` `;` tab `|`),\n  defaulting to `,`.\n- Normalize cells into deterministic stripped string values.\n- Expose row extraction helpers mirroring `XlsxParser.rows`.\n

    Constraints:

    - All values are strings; consumers requiring typed values must\n  convert on their side.\n- Quoted fields containing delimiters/newlines are handled by\n  the standard ``csv`` module.\n

    Initialize the parser.

    Parameters:

    Name Type Description Default content Content

    CSV content to parse; its type must be supported.

    required"},{"location":"csv/#omniread.csv.CsvParser-attributes","title":"Attributes","text":""},{"location":"csv/#omniread.csv.CsvParser.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {CSV}\n

    Set of content types supported by this parser (CSV only).

    "},{"location":"csv/#omniread.csv.CsvParser-functions","title":"Functions","text":""},{"location":"csv/#omniread.csv.CsvParser.parse","title":"parse","text":"
    parse() -> list[list[str]]\n

    Parse the document into normalized string rows.

    Returns:

    Type Description list[list[str]]

    list[list[str]]: Rows of the document.

    "},{"location":"csv/#omniread.csv.CsvParser.rows","title":"rows","text":"
    rows(*, skip_empty: bool = True) -> list[list[str]]\n

    Extract normalized string rows from the document.

    Parameters:

    Name Type Description Default skip_empty bool

    When True (default), rows whose cells are all blank are omitted.

    True

    Returns:

    Type Description list[list[str]]

    list[list[str]]: Normalized rows; trailing blank cells are trimmed per row.

    "},{"location":"csv/#omniread.csv.CsvParser.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"csv/#omniread.csv.CsvParserBase","title":"CsvParserBase","text":"
    CsvParserBase(content: Content)\n

    Bases: BaseParser[T], Generic[T]

    Base csv parser.

    Notes

    Responsibilities:

    - This class enforces csv content-type compatibility and provides\n  the extension point for implementing concrete csv parsing\n  strategies.\n

    Constraints:

    - Concrete implementations must define the output type `T` and\n  implement the `parse()` method.\n

    Initialize the parser with content to be parsed.

    Parameters:

    Name Type Description Default content Content

    Content instance to be parsed.

    required

    Raises:

    Type Description ValueError

    If the content type is not supported by this parser.

    "},{"location":"csv/#omniread.csv.CsvParserBase-attributes","title":"Attributes","text":""},{"location":"csv/#omniread.csv.CsvParserBase.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {CSV}\n

    Set of content types supported by this parser (CSV only).

    "},{"location":"csv/#omniread.csv.CsvParserBase-functions","title":"Functions","text":""},{"location":"csv/#omniread.csv.CsvParserBase.parse","title":"parse abstractmethod","text":"
    parse() -> T\n

    Parse csv content into a structured output.

    Returns:

    Name Type Description T T

    Parsed representation of type T.

    Raises:

    Type Description Exception

    Parsing-specific errors as defined by the implementation.

    "},{"location":"csv/#omniread.csv.CsvParserBase.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"csv/#omniread.csv.CsvScraper","title":"CsvScraper","text":"
    CsvScraper(*, client: BaseCsvClient)\n

    Scraper for csv documents.

    Notes

    Responsibilities:

    - Fetch raw csv bytes via the configured client.\n- Wrap the payload in a canonical `Content` instance with the\n  CSV content type and source identifier.\n

    Constraints:

    - The scraper does not perform parsing or interpretation.\n- Does not assume a specific storage backend.\n

    Initialize the CSV scraper.

    Parameters:

    Name Type Description Default client BaseCsvClient

    Client responsible for retrieving raw csv bytes.

    required"},{"location":"csv/#omniread.csv.CsvScraper-functions","title":"Functions","text":""},{"location":"csv/#omniread.csv.CsvScraper.fetch","title":"fetch","text":"
    fetch(\n    source: Any,\n    *,\n    metadata: Mapping[str, Any] | None = None\n) -> Content\n

    Fetch a csv document from the given source.

    Parameters:

    Name Type Description Default source Any

    Identifier of the csv source as understood by the configured client.

    required metadata Mapping[str, Any] | None

    Optional metadata to attach to the returned content.

    None

    Returns:

    Name Type Description Content Content

    A Content instance containing raw csv bytes, source identifier, CSV content type, and optional metadata.

    Raises:

    Type Description Exception

    Retrieval-specific errors raised by the client.

    "},{"location":"csv/#omniread.csv.FileSystemCsvClient","title":"FileSystemCsvClient","text":"

    Bases: BaseCsvClient

    CSV client that reads from the local filesystem.

    Notes

    Guarantees:

    - This client reads csv files directly from the disk and\n  returns their raw binary contents.\n
    "},{"location":"csv/#omniread.csv.FileSystemCsvClient-functions","title":"Functions","text":""},{"location":"csv/#omniread.csv.FileSystemCsvClient.fetch","title":"fetch","text":"
    fetch(path: Path) -> bytes\n

    Read a csv file from the local filesystem.

    Parameters:

    Name Type Description Default path Path

    Filesystem path to the csv file.

    required

    Returns:

    Name Type Description bytes bytes

    Raw csv bytes.

    Raises:

    Type Description FileNotFoundError

    If the path does not exist.

    ValueError

    If the path exists but is not a file.

    "},{"location":"csv/client/","title":"Client","text":""},{"location":"csv/client/#omniread.csv.client","title":"omniread.csv.client","text":""},{"location":"csv/client/#omniread.csv.client--summary","title":"Summary","text":"

    CSV client abstractions for OmniRead.

    This module defines the client layer responsible for retrieving raw comma-separated-value document bytes from a concrete backing store.

    Clients provide low-level access to csv binaries and are intentionally decoupled from scraping and parsing logic. They do not perform validation, interpretation, or content extraction.

    Typical backing stores include:

    • Local filesystems
    • Object storage (S3, GCS, etc.)
    • Network file systems
    "},{"location":"csv/client/#omniread.csv.client-classes","title":"Classes","text":""},{"location":"csv/client/#omniread.csv.client.BaseCsvClient","title":"BaseCsvClient","text":"

    Bases: ABC

    Abstract client responsible for retrieving csv bytes.

    Retrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).

    Notes

    Responsibilities:

    - Implementations must accept a source identifier appropriate to\n  the backing store.\n- Return the full csv binary payload.\n- Raise retrieval-specific errors on failure.\n
    "},{"location":"csv/client/#omniread.csv.client.BaseCsvClient-functions","title":"Functions","text":""},{"location":"csv/client/#omniread.csv.client.BaseCsvClient.fetch","title":"fetch abstractmethod","text":"
    fetch(source: Any) -> bytes\n

    Fetch raw csv bytes from the given source.

    Parameters:

    Name Type Description Default source Any

    Identifier of the csv location, such as a file path, object storage key, or remote reference.

    required

    Returns:

    Name Type Description bytes bytes

    Raw csv bytes.

    Raises:

    Type Description Exception

    Retrieval-specific errors defined by the implementation.

    "},{"location":"csv/client/#omniread.csv.client.FileSystemCsvClient","title":"FileSystemCsvClient","text":"

    Bases: BaseCsvClient

    CSV client that reads from the local filesystem.

    Notes

    Guarantees:

    - This client reads csv files directly from the disk and\n  returns their raw binary contents.\n
    "},{"location":"csv/client/#omniread.csv.client.FileSystemCsvClient-functions","title":"Functions","text":""},{"location":"csv/client/#omniread.csv.client.FileSystemCsvClient.fetch","title":"fetch","text":"
    fetch(path: Path) -> bytes\n

    Read a csv file from the local filesystem.

    Parameters:

    Name Type Description Default path Path

    Filesystem path to the csv file.

    required

    Returns:

    Name Type Description bytes bytes

    Raw csv bytes.

    Raises:

    Type Description FileNotFoundError

    If the path does not exist.

    ValueError

    If the path exists but is not a file.

    "},{"location":"csv/parser/","title":"Parser","text":""},{"location":"csv/parser/#omniread.csv.parser","title":"omniread.csv.parser","text":""},{"location":"csv/parser/#omniread.csv.parser--summary","title":"Summary","text":"

    CSV parser implementations for OmniRead.

    This module provides a concrete, generic parser for comma-separated-value documents. It exposes records as lists of string cells so downstream consumers can interpret tabular content without depending on the csv module directly.

    The parser is intentionally statement-agnostic: it performs no header detection or column interpretation beyond basic cell normalization and delimiter detection.

    "},{"location":"csv/parser/#omniread.csv.parser-classes","title":"Classes","text":""},{"location":"csv/parser/#omniread.csv.parser.CsvParser","title":"CsvParser","text":"
    CsvParser(content: Content)\n

    Bases: CsvParserBase[list[list[str]]]

    Generic csv parser producing string rows from the document.

    Notes

    Responsibilities:

    - Decode the payload (UTF-8 with BOM support, Latin-1 fallback).\n- Detect the delimiter from a leading sample (`,` `;` tab `|`),\n  defaulting to `,`.\n- Normalize cells into deterministic stripped string values.\n- Expose row extraction helpers mirroring `XlsxParser.rows`.\n

    Constraints:

    - All values are strings; consumers requiring typed values must\n  convert on their side.\n- Quoted fields containing delimiters/newlines are handled by\n  the standard ``csv`` module.\n

    Initialize the parser.

    Parameters:

    Name Type Description Default content Content

    CSV content to parse; its type must be supported.

    required"},{"location":"csv/parser/#omniread.csv.parser.CsvParser-attributes","title":"Attributes","text":""},{"location":"csv/parser/#omniread.csv.parser.CsvParser.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {CSV}\n

    Set of content types supported by this parser (CSV only).

    "},{"location":"csv/parser/#omniread.csv.parser.CsvParser-functions","title":"Functions","text":""},{"location":"csv/parser/#omniread.csv.parser.CsvParser.parse","title":"parse","text":"
    parse() -> list[list[str]]\n

    Parse the document into normalized string rows.

    Returns:

    Type Description list[list[str]]

    list[list[str]]: Rows of the document.

    "},{"location":"csv/parser/#omniread.csv.parser.CsvParser.rows","title":"rows","text":"
    rows(*, skip_empty: bool = True) -> list[list[str]]\n

    Extract normalized string rows from the document.

    Parameters:

    Name Type Description Default skip_empty bool

    When True (default), rows whose cells are all blank are omitted.

    True

    Returns:

    Type Description list[list[str]]

    list[list[str]]: Normalized rows; trailing blank cells are trimmed per row.

    "},{"location":"csv/parser/#omniread.csv.parser.CsvParser.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"csv/parser_base/","title":"Parser Base","text":""},{"location":"csv/parser_base/#omniread.csv.parser_base","title":"omniread.csv.parser_base","text":""},{"location":"csv/parser_base/#omniread.csv.parser_base--summary","title":"Summary","text":"

    CSV parser base implementation for OmniRead.

    This module defines the CSV-specific parser contract, extending the format-agnostic BaseParser with constraints appropriate for comma-separated-value documents.

    "},{"location":"csv/parser_base/#omniread.csv.parser_base-classes","title":"Classes","text":""},{"location":"csv/parser_base/#omniread.csv.parser_base.CsvParserBase","title":"CsvParserBase","text":"
    CsvParserBase(content: Content)\n

    Bases: BaseParser[T], Generic[T]

    Base csv parser.

    Notes

    Responsibilities:

    - This class enforces csv content-type compatibility and provides\n  the extension point for implementing concrete csv parsing\n  strategies.\n

    Constraints:

    - Concrete implementations must define the output type `T` and\n  implement the `parse()` method.\n

    Initialize the parser with content to be parsed.

    Parameters:

    Name Type Description Default content Content

    Content instance to be parsed.

    required

    Raises:

    Type Description ValueError

    If the content type is not supported by this parser.

    "},{"location":"csv/parser_base/#omniread.csv.parser_base.CsvParserBase-attributes","title":"Attributes","text":""},{"location":"csv/parser_base/#omniread.csv.parser_base.CsvParserBase.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {CSV}\n

    Set of content types supported by this parser (CSV only).

    "},{"location":"csv/parser_base/#omniread.csv.parser_base.CsvParserBase-functions","title":"Functions","text":""},{"location":"csv/parser_base/#omniread.csv.parser_base.CsvParserBase.parse","title":"parse abstractmethod","text":"
    parse() -> T\n

    Parse csv content into a structured output.

    Returns:

    Name Type Description T T

    Parsed representation of type T.

    Raises:

    Type Description Exception

    Parsing-specific errors as defined by the implementation.

    "},{"location":"csv/parser_base/#omniread.csv.parser_base.CsvParserBase.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"csv/scraper/","title":"Scraper","text":""},{"location":"csv/scraper/#omniread.csv.scraper","title":"omniread.csv.scraper","text":""},{"location":"csv/scraper/#omniread.csv.scraper--summary","title":"Summary","text":"

    CSV scraper for OmniRead.

    This module defines the scraper responsible for acquiring raw comma-separated-value document content from a backing store via a configured client.

    The scraper does not interpret or parse the acquired bytes; it wraps them in the canonical Content model.

    "},{"location":"csv/scraper/#omniread.csv.scraper-classes","title":"Classes","text":""},{"location":"csv/scraper/#omniread.csv.scraper.CsvScraper","title":"CsvScraper","text":"
    CsvScraper(*, client: BaseCsvClient)\n

    Scraper for csv documents.

    Notes

    Responsibilities:

    - Fetch raw csv bytes via the configured client.\n- Wrap the payload in a canonical `Content` instance with the\n  CSV content type and source identifier.\n

    Constraints:

    - The scraper does not perform parsing or interpretation.\n- Does not assume a specific storage backend.\n

    Initialize the CSV scraper.

    Parameters:

    Name Type Description Default client BaseCsvClient

    Client responsible for retrieving raw csv bytes.

    required"},{"location":"csv/scraper/#omniread.csv.scraper.CsvScraper-functions","title":"Functions","text":""},{"location":"csv/scraper/#omniread.csv.scraper.CsvScraper.fetch","title":"fetch","text":"
    fetch(\n    source: Any,\n    *,\n    metadata: Mapping[str, Any] | None = None\n) -> Content\n

    Fetch a csv document from the given source.

    Parameters:

    Name Type Description Default source Any

    Identifier of the csv source as understood by the configured client.

    required metadata Mapping[str, Any] | None

    Optional metadata to attach to the returned content.

    None

    Returns:

    Name Type Description Content Content

    A Content instance containing raw csv bytes, source identifier, CSV content type, and optional metadata.

    Raises:

    Type Description Exception

    Retrieval-specific errors raised by the client.

    "},{"location":"html/","title":"Html","text":""},{"location":"html/#omniread.html","title":"omniread.html","text":""},{"location":"html/#omniread.html--summary","title":"Summary","text":"

    HTML format implementation for OmniRead.

    This package provides HTML-specific implementations of the core OmniRead contracts defined in omniread.core.

    It includes:

    • HTML parsers that interpret HTML content.
    • HTML scrapers that retrieve HTML documents.

    Key characteristics:

    • Implements, but does not redefine, core contracts.
    • May contain HTML-specific behavior and edge-case handling.
    • Produces canonical content models defined in omniread.core.content.

    Consumers should depend on omniread.core interfaces wherever possible and use this package only when HTML-specific behavior is required.

    "},{"location":"html/#omniread.html--public-api","title":"Public API","text":"
    • HTMLScraper
    • HTMLParser
    "},{"location":"html/#omniread.html-classes","title":"Classes","text":""},{"location":"html/#omniread.html.HTMLParser","title":"HTMLParser","text":"
    HTMLParser(content: Content, features: str = 'html.parser')\n

    Bases: BaseParser[T], Generic[T]

    Base HTML parser.

    Notes

    Responsibilities:

    - This class extends the core `BaseParser` with HTML-specific behavior,\n  including DOM parsing via BeautifulSoup and reusable extraction helpers.\n- Provides reusable helpers for HTML extraction. Concrete parsers must\n  explicitly define the return type.\n

    Guarantees:

    - Accepts only HTML content.\n- Owns a parsed BeautifulSoup DOM tree.\n- Provides pure helper utilities for common HTML structures.\n

    Constraints:

    - Concrete subclasses must define the output type `T` and implement\n  the `parse()` method.\n

    Initialize the HTML parser.

    Parameters:

    Name Type Description Default content Content

    HTML content to be parsed.

    required features str

    BeautifulSoup parser backend to use (e.g., 'html.parser', 'lxml').

    'html.parser'

    Raises:

    Type Description ValueError

    If the content is empty or not valid HTML.

    "},{"location":"html/#omniread.html.HTMLParser-attributes","title":"Attributes","text":""},{"location":"html/#omniread.html.HTMLParser.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {HTML}\n

    Set of content types supported by this parser (HTML only).

    "},{"location":"html/#omniread.html.HTMLParser-functions","title":"Functions","text":""},{"location":"html/#omniread.html.HTMLParser.parse","title":"parse abstractmethod","text":"
    parse() -> T\n

    Fully parse the HTML content into structured output.

    Returns:

    Name Type Description T T

    Parsed representation of type T.

    Notes

    Responsibilities:

    - Implementations must fully interpret the HTML DOM and return a\n  deterministic, structured output.\n
    "},{"location":"html/#omniread.html.HTMLParser.parse_div","title":"parse_div staticmethod","text":"
    parse_div(div: Tag, *, separator: str = ' ') -> str\n

    Extract normalized text from a <div> element.

    Parameters:

    Name Type Description Default div Tag

    BeautifulSoup tag representing a <div>.

    required separator str

    String used to separate text nodes.

    ' '

    Returns:

    Name Type Description str str

    Flattened, whitespace-normalized text content.

    "},{"location":"html/#omniread.html.HTMLParser.parse_link","title":"parse_link staticmethod","text":"
    parse_link(a: Tag) -> str | None\n

    Extract the hyperlink reference from an <a> element.

    Parameters:

    Name Type Description Default a Tag

    BeautifulSoup tag representing an anchor.

    required

    Returns:

    Type Description str | None

    str | None: The value of the href attribute, or None if absent.

    "},{"location":"html/#omniread.html.HTMLParser.parse_meta","title":"parse_meta","text":"
    parse_meta() -> dict[str, Any]\n

    Extract high-level metadata from the HTML document.

    Returns:

    Type Description dict[str, Any]

    dict[str, Any]: Dictionary containing extracted metadata.

    Notes

    Responsibilities:

    - Extract high-level metadata from the HTML document.\n- This includes: Document title, `<meta>` tag name/property to\n  content mappings.\n
    "},{"location":"html/#omniread.html.HTMLParser.parse_table","title":"parse_table staticmethod","text":"
    parse_table(table: Tag) -> list[list[str]]\n

    Parse an HTML table into a 2D list of strings.

    Parameters:

    Name Type Description Default table Tag

    BeautifulSoup tag representing a <table>.

    required

    Returns:

    Type Description list[list[str]]

    list[list[str]]: A list of rows, where each row is a list of cell text values.

    "},{"location":"html/#omniread.html.HTMLParser.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"html/#omniread.html.HTMLScraper","title":"HTMLScraper","text":"
    HTMLScraper(\n    *,\n    client: httpx.Client | None = None,\n    timeout: float = 15.0,\n    headers: Mapping[str, str] | None = None,\n    follow_redirects: bool = True\n)\n

    Bases: BaseScraper

    Base HTML scraper using httpx.

    Notes

    Responsibilities:

    - This scraper retrieves HTML documents over HTTP(S) and returns\n  them as raw content wrapped in a `Content` object.\n- Fetches raw bytes and metadata only.\n- The scraper uses `httpx.Client` for HTTP requests, enforces an\n  HTML content type, and preserves HTTP response metadata.\n

    Constraints:

    - The scraper does not: Parse HTML, perform retries or backoff,\n  handle non-HTML responses.\n

    Initialize the HTML scraper.

    Parameters:

    Name Type Description Default client Client | None

    Optional pre-configured httpx.Client. If omitted, a client is created internally.

    None timeout float

    Request timeout in seconds.

    15.0 headers Mapping[str, str] | None

    Optional default HTTP headers.

    None follow_redirects bool

    Whether to follow HTTP redirects.

    True"},{"location":"html/#omniread.html.HTMLScraper-functions","title":"Functions","text":""},{"location":"html/#omniread.html.HTMLScraper.fetch","title":"fetch","text":"
    fetch(\n    source: str,\n    *,\n    metadata: Mapping[str, Any] | None = None\n) -> Content\n

    Fetch an HTML document from the given source.

    Parameters:

    Name Type Description Default source str

    URL of the HTML document.

    required metadata Mapping[str, Any] | None

    Optional metadata to be merged into the returned content.

    None

    Returns:

    Name Type Description Content Content

    A Content instance containing raw HTML bytes, source URL, HTML content type, and HTTP response metadata.

    Raises:

    Type Description HTTPError

    If the HTTP request fails.

    ValueError

    If the response is not valid HTML.

    "},{"location":"html/#omniread.html.HTMLScraper.validate_content_type","title":"validate_content_type","text":"
    validate_content_type(response: httpx.Response) -> None\n

    Validate that the HTTP response contains HTML content.

    Parameters:

    Name Type Description Default response Response

    HTTP response returned by httpx.

    required

    Raises:

    Type Description ValueError

    If the Content-Type header is missing or does not indicate HTML content.

    "},{"location":"html/parser/","title":"Parser","text":""},{"location":"html/parser/#omniread.html.parser","title":"omniread.html.parser","text":""},{"location":"html/parser/#omniread.html.parser--summary","title":"Summary","text":"

    HTML parser base implementations for OmniRead.

    This module provides reusable HTML parsing utilities built on top of the abstract parser contracts defined in omniread.core.parser.

    It supplies:

    • Content-type enforcement for HTML inputs
    • BeautifulSoup initialization and lifecycle management
    • Common helper methods for extracting structured data from HTML elements

    Concrete parsers must subclass HTMLParser and implement the parse() method to return a structured representation appropriate for their use case.

    "},{"location":"html/parser/#omniread.html.parser-classes","title":"Classes","text":""},{"location":"html/parser/#omniread.html.parser.HTMLParser","title":"HTMLParser","text":"
    HTMLParser(content: Content, features: str = 'html.parser')\n

    Bases: BaseParser[T], Generic[T]

    Base HTML parser.

    Notes

    Responsibilities:

    - This class extends the core `BaseParser` with HTML-specific behavior,\n  including DOM parsing via BeautifulSoup and reusable extraction helpers.\n- Provides reusable helpers for HTML extraction. Concrete parsers must\n  explicitly define the return type.\n

    Guarantees:

    - Accepts only HTML content.\n- Owns a parsed BeautifulSoup DOM tree.\n- Provides pure helper utilities for common HTML structures.\n

    Constraints:

    - Concrete subclasses must define the output type `T` and implement\n  the `parse()` method.\n

    Initialize the HTML parser.

    Parameters:

    Name Type Description Default content Content

    HTML content to be parsed.

    required features str

    BeautifulSoup parser backend to use (e.g., 'html.parser', 'lxml').

    'html.parser'

    Raises:

    Type Description ValueError

    If the content is empty or not valid HTML.

    "},{"location":"html/parser/#omniread.html.parser.HTMLParser-attributes","title":"Attributes","text":""},{"location":"html/parser/#omniread.html.parser.HTMLParser.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {HTML}\n

    Set of content types supported by this parser (HTML only).

    "},{"location":"html/parser/#omniread.html.parser.HTMLParser-functions","title":"Functions","text":""},{"location":"html/parser/#omniread.html.parser.HTMLParser.parse","title":"parse abstractmethod","text":"
    parse() -> T\n

    Fully parse the HTML content into structured output.

    Returns:

    Name Type Description T T

    Parsed representation of type T.

    Notes

    Responsibilities:

    - Implementations must fully interpret the HTML DOM and return a\n  deterministic, structured output.\n
    "},{"location":"html/parser/#omniread.html.parser.HTMLParser.parse_div","title":"parse_div staticmethod","text":"
    parse_div(div: Tag, *, separator: str = ' ') -> str\n

    Extract normalized text from a <div> element.

    Parameters:

    Name Type Description Default div Tag

    BeautifulSoup tag representing a <div>.

    required separator str

    String used to separate text nodes.

    ' '

    Returns:

    Name Type Description str str

    Flattened, whitespace-normalized text content.

    "},{"location":"html/parser/#omniread.html.parser.HTMLParser.parse_link","title":"parse_link staticmethod","text":"
    parse_link(a: Tag) -> str | None\n

    Extract the hyperlink reference from an <a> element.

    Parameters:

    Name Type Description Default a Tag

    BeautifulSoup tag representing an anchor.

    required

    Returns:

    Type Description str | None

    str | None: The value of the href attribute, or None if absent.

    "},{"location":"html/parser/#omniread.html.parser.HTMLParser.parse_meta","title":"parse_meta","text":"
    parse_meta() -> dict[str, Any]\n

    Extract high-level metadata from the HTML document.

    Returns:

    Type Description dict[str, Any]

    dict[str, Any]: Dictionary containing extracted metadata.

    Notes

    Responsibilities:

    - Extract high-level metadata from the HTML document.\n- This includes: Document title, `<meta>` tag name/property to\n  content mappings.\n
    "},{"location":"html/parser/#omniread.html.parser.HTMLParser.parse_table","title":"parse_table staticmethod","text":"
    parse_table(table: Tag) -> list[list[str]]\n

    Parse an HTML table into a 2D list of strings.

    Parameters:

    Name Type Description Default table Tag

    BeautifulSoup tag representing a <table>.

    required

    Returns:

    Type Description list[list[str]]

    list[list[str]]: A list of rows, where each row is a list of cell text values.

    "},{"location":"html/parser/#omniread.html.parser.HTMLParser.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"html/scraper/","title":"Scraper","text":""},{"location":"html/scraper/#omniread.html.scraper","title":"omniread.html.scraper","text":""},{"location":"html/scraper/#omniread.html.scraper--summary","title":"Summary","text":"

    HTML scraping implementation for OmniRead.

    This module provides an HTTP-based scraper for retrieving HTML documents. It implements the core BaseScraper contract using httpx as the transport layer.

    This scraper is responsible for:

    • Fetching raw HTML bytes over HTTP(S)
    • Validating response content type
    • Attaching HTTP metadata to the returned content

    This scraper is not responsible for:

    • Parsing or interpreting HTML
    • Retrying failed requests
    • Managing crawl policies or rate limiting
    "},{"location":"html/scraper/#omniread.html.scraper-classes","title":"Classes","text":""},{"location":"html/scraper/#omniread.html.scraper.HTMLScraper","title":"HTMLScraper","text":"
    HTMLScraper(\n    *,\n    client: httpx.Client | None = None,\n    timeout: float = 15.0,\n    headers: Mapping[str, str] | None = None,\n    follow_redirects: bool = True\n)\n

    Bases: BaseScraper

    Base HTML scraper using httpx.

    Notes

    Responsibilities:

    - This scraper retrieves HTML documents over HTTP(S) and returns\n  them as raw content wrapped in a `Content` object.\n- Fetches raw bytes and metadata only.\n- The scraper uses `httpx.Client` for HTTP requests, enforces an\n  HTML content type, and preserves HTTP response metadata.\n

    Constraints:

    - The scraper does not: Parse HTML, perform retries or backoff,\n  handle non-HTML responses.\n

    Initialize the HTML scraper.

    Parameters:

    Name Type Description Default client Client | None

    Optional pre-configured httpx.Client. If omitted, a client is created internally.

    None timeout float

    Request timeout in seconds.

    15.0 headers Mapping[str, str] | None

    Optional default HTTP headers.

    None follow_redirects bool

    Whether to follow HTTP redirects.

    True"},{"location":"html/scraper/#omniread.html.scraper.HTMLScraper-functions","title":"Functions","text":""},{"location":"html/scraper/#omniread.html.scraper.HTMLScraper.fetch","title":"fetch","text":"
    fetch(\n    source: str,\n    *,\n    metadata: Mapping[str, Any] | None = None\n) -> Content\n

    Fetch an HTML document from the given source.

    Parameters:

    Name Type Description Default source str

    URL of the HTML document.

    required metadata Mapping[str, Any] | None

    Optional metadata to be merged into the returned content.

    None

    Returns:

    Name Type Description Content Content

    A Content instance containing raw HTML bytes, source URL, HTML content type, and HTTP response metadata.

    Raises:

    Type Description HTTPError

    If the HTTP request fails.

    ValueError

    If the response is not valid HTML.

    "},{"location":"html/scraper/#omniread.html.scraper.HTMLScraper.validate_content_type","title":"validate_content_type","text":"
    validate_content_type(response: httpx.Response) -> None\n

    Validate that the HTTP response contains HTML content.

    Parameters:

    Name Type Description Default response Response

    HTTP response returned by httpx.

    required

    Raises:

    Type Description ValueError

    If the Content-Type header is missing or does not indicate HTML content.

    "},{"location":"pdf/","title":"Pdf","text":""},{"location":"pdf/#omniread.pdf","title":"omniread.pdf","text":""},{"location":"pdf/#omniread.pdf--summary","title":"Summary","text":"

    PDF format implementation for OmniRead.

    This package provides PDF-specific implementations of the core OmniRead contracts defined in omniread.core.

    Unlike HTML, PDF handling requires an explicit client layer for document access. This package therefore includes:

    • PDF clients for acquiring raw PDF data.
    • PDF scrapers that coordinate client access.
    • PDF parsers that extract structured content from PDF binaries.

    Public exports from this package represent the supported PDF pipeline and are safe for consumers to import directly when working with PDFs.

    "},{"location":"pdf/#omniread.pdf--public-api","title":"Public API","text":"
    • FileSystemPDFClient
    • PDFScraper
    • PDFParser
    "},{"location":"pdf/#omniread.pdf-classes","title":"Classes","text":""},{"location":"pdf/#omniread.pdf.FileSystemPDFClient","title":"FileSystemPDFClient","text":"

    Bases: BasePDFClient

    PDF client that reads from the local filesystem.

    Notes

    Guarantees:

    - This client reads PDF files directly from the disk and returns\n  their raw binary contents.\n
    "},{"location":"pdf/#omniread.pdf.FileSystemPDFClient-functions","title":"Functions","text":""},{"location":"pdf/#omniread.pdf.FileSystemPDFClient.fetch","title":"fetch","text":"
    fetch(path: Path) -> bytes\n

    Read a PDF file from the local filesystem.

    Parameters:

    Name Type Description Default path Path

    Filesystem path to the PDF file.

    required

    Returns:

    Name Type Description bytes bytes

    Raw PDF bytes.

    Raises:

    Type Description FileNotFoundError

    If the path does not exist.

    ValueError

    If the path exists but is not a file.

    "},{"location":"pdf/#omniread.pdf.PDFParser","title":"PDFParser","text":"
    PDFParser(content: Content)\n

    Bases: BaseParser[T], Generic[T]

    Base PDF parser.

    Notes

    Responsibilities:

    - This class enforces PDF content-type compatibility and provides\n  the extension point for implementing concrete PDF parsing strategies.\n

    Constraints:

    - Concrete implementations must define the output type `T` and\n  implement the `parse()` method.\n

    Initialize the parser with content to be parsed.

    Parameters:

    Name Type Description Default content Content

    Content instance to be parsed.

    required

    Raises:

    Type Description ValueError

    If the content type is not supported by this parser.

    "},{"location":"pdf/#omniread.pdf.PDFParser-attributes","title":"Attributes","text":""},{"location":"pdf/#omniread.pdf.PDFParser.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {PDF}\n

    Set of content types supported by this parser (PDF only).

    "},{"location":"pdf/#omniread.pdf.PDFParser-functions","title":"Functions","text":""},{"location":"pdf/#omniread.pdf.PDFParser.parse","title":"parse abstractmethod","text":"
    parse() -> T\n

    Parse PDF content into a structured output.

    Returns:

    Name Type Description T T

    Parsed representation of type T.

    Raises:

    Type Description Exception

    Parsing-specific errors as defined by the implementation.

    Notes

    Responsibilities:

    - Implementations must fully interpret the PDF binary payload and\n  return a deterministic, structured output.\n
    "},{"location":"pdf/#omniread.pdf.PDFParser.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"pdf/#omniread.pdf.PDFScraper","title":"PDFScraper","text":"
    PDFScraper(*, client: BasePDFClient)\n

    Bases: BaseScraper

    Scraper for PDF sources.

    Notes

    Responsibilities:

    - Delegates byte retrieval to a PDF client and normalizes output\n  into `Content`.\n- Preserves caller-provided metadata.\n

    Constraints:

    - The scraper does not perform parsing or interpretation.\n- Does not assume a specific storage backend.\n

    Initialize the PDF scraper.

    Parameters:

    Name Type Description Default client BasePDFClient

    PDF client responsible for retrieving raw PDF bytes.

    required"},{"location":"pdf/#omniread.pdf.PDFScraper-functions","title":"Functions","text":""},{"location":"pdf/#omniread.pdf.PDFScraper.fetch","title":"fetch","text":"
    fetch(\n    source: Any,\n    *,\n    metadata: Mapping[str, Any] | None = None\n) -> Content\n

    Fetch a PDF document from the given source.

    Parameters:

    Name Type Description Default source Any

    Identifier of the PDF source as understood by the configured PDF client.

    required metadata Mapping[str, Any] | None

    Optional metadata to attach to the returned content.

    None

    Returns:

    Name Type Description Content Content

    A Content instance containing raw PDF bytes, source identifier, PDF content type, and optional metadata.

    Raises:

    Type Description Exception

    Retrieval-specific errors raised by the PDF client.

    "},{"location":"pdf/client/","title":"Client","text":""},{"location":"pdf/client/#omniread.pdf.client","title":"omniread.pdf.client","text":""},{"location":"pdf/client/#omniread.pdf.client--summary","title":"Summary","text":"

    PDF client abstractions for OmniRead.

    This module defines the client layer responsible for retrieving raw PDF bytes from a concrete backing store.

    Clients provide low-level access to PDF binaries and are intentionally decoupled from scraping and parsing logic. They do not perform validation, interpretation, or content extraction.

    Typical backing stores include:

    • Local filesystems
    • Object storage (S3, GCS, etc.)
    • Network file systems
    "},{"location":"pdf/client/#omniread.pdf.client-classes","title":"Classes","text":""},{"location":"pdf/client/#omniread.pdf.client.BasePDFClient","title":"BasePDFClient","text":"

    Bases: ABC

    Abstract client responsible for retrieving PDF bytes.

    Retrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).

    Notes

    Responsibilities:

    - Implementations must accept a source identifier appropriate to\n  the backing store.\n- Return the full PDF binary payload.\n- Raise retrieval-specific errors on failure.\n
    "},{"location":"pdf/client/#omniread.pdf.client.BasePDFClient-functions","title":"Functions","text":""},{"location":"pdf/client/#omniread.pdf.client.BasePDFClient.fetch","title":"fetch abstractmethod","text":"
    fetch(source: Any) -> bytes\n

    Fetch raw PDF bytes from the given source.

    Parameters:

    Name Type Description Default source Any

    Identifier of the PDF location, such as a file path, object storage key, or remote reference.

    required

    Returns:

    Name Type Description bytes bytes

    Raw PDF bytes.

    Raises:

    Type Description Exception

    Retrieval-specific errors defined by the implementation.

    "},{"location":"pdf/client/#omniread.pdf.client.FileSystemPDFClient","title":"FileSystemPDFClient","text":"

    Bases: BasePDFClient

    PDF client that reads from the local filesystem.

    Notes

    Guarantees:

    - This client reads PDF files directly from the disk and returns\n  their raw binary contents.\n
    "},{"location":"pdf/client/#omniread.pdf.client.FileSystemPDFClient-functions","title":"Functions","text":""},{"location":"pdf/client/#omniread.pdf.client.FileSystemPDFClient.fetch","title":"fetch","text":"
    fetch(path: Path) -> bytes\n

    Read a PDF file from the local filesystem.

    Parameters:

    Name Type Description Default path Path

    Filesystem path to the PDF file.

    required

    Returns:

    Name Type Description bytes bytes

    Raw PDF bytes.

    Raises:

    Type Description FileNotFoundError

    If the path does not exist.

    ValueError

    If the path exists but is not a file.

    "},{"location":"pdf/parser/","title":"Parser","text":""},{"location":"pdf/parser/#omniread.pdf.parser","title":"omniread.pdf.parser","text":""},{"location":"pdf/parser/#omniread.pdf.parser--summary","title":"Summary","text":"

    PDF parser base implementations for OmniRead.

    This module defines the PDF-specific parser contract, extending the format-agnostic BaseParser with constraints appropriate for PDF content.

    PDF parsers are responsible for interpreting binary PDF data and producing structured representations suitable for downstream consumption.

    "},{"location":"pdf/parser/#omniread.pdf.parser-classes","title":"Classes","text":""},{"location":"pdf/parser/#omniread.pdf.parser.PDFParser","title":"PDFParser","text":"
    PDFParser(content: Content)\n

    Bases: BaseParser[T], Generic[T]

    Base PDF parser.

    Notes

    Responsibilities:

    - This class enforces PDF content-type compatibility and provides\n  the extension point for implementing concrete PDF parsing strategies.\n

    Constraints:

    - Concrete implementations must define the output type `T` and\n  implement the `parse()` method.\n

    Initialize the parser with content to be parsed.

    Parameters:

    Name Type Description Default content Content

    Content instance to be parsed.

    required

    Raises:

    Type Description ValueError

    If the content type is not supported by this parser.

    "},{"location":"pdf/parser/#omniread.pdf.parser.PDFParser-attributes","title":"Attributes","text":""},{"location":"pdf/parser/#omniread.pdf.parser.PDFParser.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {PDF}\n

    Set of content types supported by this parser (PDF only).

    "},{"location":"pdf/parser/#omniread.pdf.parser.PDFParser-functions","title":"Functions","text":""},{"location":"pdf/parser/#omniread.pdf.parser.PDFParser.parse","title":"parse abstractmethod","text":"
    parse() -> T\n

    Parse PDF content into a structured output.

    Returns:

    Name Type Description T T

    Parsed representation of type T.

    Raises:

    Type Description Exception

    Parsing-specific errors as defined by the implementation.

    Notes

    Responsibilities:

    - Implementations must fully interpret the PDF binary payload and\n  return a deterministic, structured output.\n
    "},{"location":"pdf/parser/#omniread.pdf.parser.PDFParser.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"pdf/scraper/","title":"Scraper","text":""},{"location":"pdf/scraper/#omniread.pdf.scraper","title":"omniread.pdf.scraper","text":""},{"location":"pdf/scraper/#omniread.pdf.scraper--summary","title":"Summary","text":"

    PDF scraping implementation for OmniRead.

    This module provides a PDF-specific scraper that coordinates PDF byte retrieval via a client and normalizes the result into a Content object.

    The scraper implements the core BaseScraper contract while delegating all storage and access concerns to a BasePDFClient implementation.

    "},{"location":"pdf/scraper/#omniread.pdf.scraper-classes","title":"Classes","text":""},{"location":"pdf/scraper/#omniread.pdf.scraper.PDFScraper","title":"PDFScraper","text":"
    PDFScraper(*, client: BasePDFClient)\n

    Bases: BaseScraper

    Scraper for PDF sources.

    Notes

    Responsibilities:

    - Delegates byte retrieval to a PDF client and normalizes output\n  into `Content`.\n- Preserves caller-provided metadata.\n

    Constraints:

    - The scraper does not perform parsing or interpretation.\n- Does not assume a specific storage backend.\n

    Initialize the PDF scraper.

    Parameters:

    Name Type Description Default client BasePDFClient

    PDF client responsible for retrieving raw PDF bytes.

    required"},{"location":"pdf/scraper/#omniread.pdf.scraper.PDFScraper-functions","title":"Functions","text":""},{"location":"pdf/scraper/#omniread.pdf.scraper.PDFScraper.fetch","title":"fetch","text":"
    fetch(\n    source: Any,\n    *,\n    metadata: Mapping[str, Any] | None = None\n) -> Content\n

    Fetch a PDF document from the given source.

    Parameters:

    Name Type Description Default source Any

    Identifier of the PDF source as understood by the configured PDF client.

    required metadata Mapping[str, Any] | None

    Optional metadata to attach to the returned content.

    None

    Returns:

    Name Type Description Content Content

    A Content instance containing raw PDF bytes, source identifier, PDF content type, and optional metadata.

    Raises:

    Type Description Exception

    Retrieval-specific errors raised by the PDF client.

    "},{"location":"xlsx/","title":"Xlsx","text":""},{"location":"xlsx/#omniread.xlsx","title":"omniread.xlsx","text":""},{"location":"xlsx/#omniread.xlsx--summary","title":"Summary","text":"

    XLSX subpackage for OmniRead.

    Provides acquisition and parsing of Office Open XML spreadsheet (xlsx) content:

    • BaseXlsxClient: abstract backing-store client for xlsx bytes.
    • FileSystemXlsxClient: local filesystem implementation.
    • XlsxScraper: wraps fetched bytes into canonical Content.
    • XlsxParserBase: content-type-enforcing parser contract.
    • XlsxParser: generic string-row parser built on openpyxl.
    "},{"location":"xlsx/#omniread.xlsx-classes","title":"Classes","text":""},{"location":"xlsx/#omniread.xlsx.BaseXlsxClient","title":"BaseXlsxClient","text":"

    Bases: ABC

    Abstract client responsible for retrieving spreadsheet bytes.

    Retrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).

    Notes

    Responsibilities:

    - Implementations must accept a source identifier appropriate to\n  the backing store.\n- Return the full xlsx binary payload.\n- Raise retrieval-specific errors on failure.\n
    "},{"location":"xlsx/#omniread.xlsx.BaseXlsxClient-functions","title":"Functions","text":""},{"location":"xlsx/#omniread.xlsx.BaseXlsxClient.fetch","title":"fetch abstractmethod","text":"
    fetch(source: Any) -> bytes\n

    Fetch raw xlsx bytes from the given source.

    Parameters:

    Name Type Description Default source Any

    Identifier of the spreadsheet location, such as a file path, object storage key, or remote reference.

    required

    Returns:

    Name Type Description bytes bytes

    Raw xlsx bytes.

    Raises:

    Type Description Exception

    Retrieval-specific errors defined by the implementation.

    "},{"location":"xlsx/#omniread.xlsx.FileSystemXlsxClient","title":"FileSystemXlsxClient","text":"

    Bases: BaseXlsxClient

    XLSX client that reads from the local filesystem.

    Notes

    Guarantees:

    - This client reads spreadsheet files directly from the disk and\n  returns their raw binary contents.\n
    "},{"location":"xlsx/#omniread.xlsx.FileSystemXlsxClient-functions","title":"Functions","text":""},{"location":"xlsx/#omniread.xlsx.FileSystemXlsxClient.fetch","title":"fetch","text":"
    fetch(path: Path) -> bytes\n

    Read an xlsx file from the local filesystem.

    Parameters:

    Name Type Description Default path Path

    Filesystem path to the spreadsheet file.

    required

    Returns:

    Name Type Description bytes bytes

    Raw xlsx bytes.

    Raises:

    Type Description FileNotFoundError

    If the path does not exist.

    ValueError

    If the path exists but is not a file.

    "},{"location":"xlsx/#omniread.xlsx.XlsxParser","title":"XlsxParser","text":"
    XlsxParser(\n    content: Content,\n    *,\n    data_only: bool = True,\n    read_only: bool = True\n)\n

    Bases: XlsxParserBase[list[list[str]]]

    Generic xlsx parser producing string rows from a worksheet.

    Notes

    Responsibilities:

    - Lazily load the workbook owned by the parser's content.\n- Normalize cells (including dates and numeric values) into\n  deterministic string representations.\n- Expose sheet discovery and row extraction helpers.\n

    Constraints:

    - Cells are rendered with ``str(value)`` after trimming; date and\n  datetime values are rendered in ISO format. Consumers requiring\n  locale-specific formatting must convert on their side.\n

    Initialize the parser.

    Parameters:

    Name Type Description Default content Content

    XLSX content to parse; its type must be supported.

    required data_only bool

    Passed to openpyxl: when True, formula cells yield their last computed value instead of the formula string.

    True read_only bool

    Passed to openpyxl: streaming mode for lower memory usage.

    True"},{"location":"xlsx/#omniread.xlsx.XlsxParser-attributes","title":"Attributes","text":""},{"location":"xlsx/#omniread.xlsx.XlsxParser.sheet_names","title":"sheet_names property","text":"
    sheet_names: list[str]\n

    Names of all worksheets contained in the workbook.

    "},{"location":"xlsx/#omniread.xlsx.XlsxParser.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {XLSX}\n

    Set of content types supported by this parser (XLSX only).

    "},{"location":"xlsx/#omniread.xlsx.XlsxParser.workbook","title":"workbook property","text":"
    workbook: Workbook\n

    The lazily loaded workbook backing this parser's content.

    "},{"location":"xlsx/#omniread.xlsx.XlsxParser-functions","title":"Functions","text":""},{"location":"xlsx/#omniread.xlsx.XlsxParser.parse","title":"parse","text":"
    parse() -> list[list[str]]\n

    Parse the first worksheet into normalized string rows.

    Returns:

    Type Description list[list[str]]

    list[list[str]]: Rows of the default (first) worksheet.

    "},{"location":"xlsx/#omniread.xlsx.XlsxParser.rows","title":"rows","text":"
    rows(\n    sheet: int | str | None = None,\n    *,\n    skip_empty: bool = True\n) -> list[list[str]]\n

    Extract normalized string rows from a worksheet.

    Parameters:

    Name Type Description Default sheet int | str | None

    Worksheet index or title; defaults to the first worksheet.

    None skip_empty bool

    When True (default), rows whose cells are all blank are omitted.

    True

    Returns:

    Type Description list[list[str]]

    list[list[str]]: Normalized rows; trailing blank cells are trimmed per row.

    Raises:

    Type Description ValueError

    If the requested sheet does not exist.

    "},{"location":"xlsx/#omniread.xlsx.XlsxParser.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"xlsx/#omniread.xlsx.XlsxParserBase","title":"XlsxParserBase","text":"
    XlsxParserBase(content: Content)\n

    Bases: BaseParser[T], Generic[T]

    Base xlsx parser.

    Notes

    Responsibilities:

    - This class enforces xlsx content-type compatibility and provides\n  the extension point for implementing concrete xlsx parsing\n  strategies.\n

    Constraints:

    - Concrete implementations must define the output type `T` and\n  implement the `parse()` method.\n

    Initialize the parser with content to be parsed.

    Parameters:

    Name Type Description Default content Content

    Content instance to be parsed.

    required

    Raises:

    Type Description ValueError

    If the content type is not supported by this parser.

    "},{"location":"xlsx/#omniread.xlsx.XlsxParserBase-attributes","title":"Attributes","text":""},{"location":"xlsx/#omniread.xlsx.XlsxParserBase.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {XLSX}\n

    Set of content types supported by this parser (XLSX only).

    "},{"location":"xlsx/#omniread.xlsx.XlsxParserBase-functions","title":"Functions","text":""},{"location":"xlsx/#omniread.xlsx.XlsxParserBase.parse","title":"parse abstractmethod","text":"
    parse() -> T\n

    Parse xlsx content into a structured output.

    Returns:

    Name Type Description T T

    Parsed representation of type T.

    Raises:

    Type Description Exception

    Parsing-specific errors as defined by the implementation.

    "},{"location":"xlsx/#omniread.xlsx.XlsxParserBase.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"xlsx/#omniread.xlsx.XlsxScraper","title":"XlsxScraper","text":"
    XlsxScraper(*, client: BaseXlsxClient)\n

    Scraper for xlsx spreadsheet documents.

    Notes

    Responsibilities:

    - Fetch raw xlsx bytes via the configured client.\n- Wrap the payload in a canonical `Content` instance with the\n  XLSX content type and source identifier.\n

    Constraints:

    - The scraper does not perform parsing or interpretation.\n- Does not assume a specific storage backend.\n

    Initialize the XLSX scraper.

    Parameters:

    Name Type Description Default client BaseXlsxClient

    Client responsible for retrieving raw spreadsheet bytes.

    required"},{"location":"xlsx/#omniread.xlsx.XlsxScraper-functions","title":"Functions","text":""},{"location":"xlsx/#omniread.xlsx.XlsxScraper.fetch","title":"fetch","text":"
    fetch(\n    source: Any,\n    *,\n    metadata: Mapping[str, Any] | None = None\n) -> Content\n

    Fetch an xlsx document from the given source.

    Parameters:

    Name Type Description Default source Any

    Identifier of the spreadsheet source as understood by the configured client.

    required metadata Mapping[str, Any] | None

    Optional metadata to attach to the returned content.

    None

    Returns:

    Name Type Description Content Content

    A Content instance containing raw xlsx bytes, source identifier, XLSX content type, and optional metadata.

    Raises:

    Type Description Exception

    Retrieval-specific errors raised by the client.

    "},{"location":"xlsx/client/","title":"Client","text":""},{"location":"xlsx/client/#omniread.xlsx.client","title":"omniread.xlsx.client","text":""},{"location":"xlsx/client/#omniread.xlsx.client--summary","title":"Summary","text":"

    XLSX client abstractions for OmniRead.

    This module defines the client layer responsible for retrieving raw Office Open XML spreadsheet bytes from a concrete backing store.

    Clients provide low-level access to xlsx binaries and are intentionally decoupled from scraping and parsing logic. They do not perform validation, interpretation, or content extraction.

    Typical backing stores include:

    • Local filesystems
    • Object storage (S3, GCS, etc.)
    • Network file systems
    "},{"location":"xlsx/client/#omniread.xlsx.client-classes","title":"Classes","text":""},{"location":"xlsx/client/#omniread.xlsx.client.BaseXlsxClient","title":"BaseXlsxClient","text":"

    Bases: ABC

    Abstract client responsible for retrieving spreadsheet bytes.

    Retrieves bytes from a specific backing store (filesystem, S3, FTP, etc.).

    Notes

    Responsibilities:

    - Implementations must accept a source identifier appropriate to\n  the backing store.\n- Return the full xlsx binary payload.\n- Raise retrieval-specific errors on failure.\n
    "},{"location":"xlsx/client/#omniread.xlsx.client.BaseXlsxClient-functions","title":"Functions","text":""},{"location":"xlsx/client/#omniread.xlsx.client.BaseXlsxClient.fetch","title":"fetch abstractmethod","text":"
    fetch(source: Any) -> bytes\n

    Fetch raw xlsx bytes from the given source.

    Parameters:

    Name Type Description Default source Any

    Identifier of the spreadsheet location, such as a file path, object storage key, or remote reference.

    required

    Returns:

    Name Type Description bytes bytes

    Raw xlsx bytes.

    Raises:

    Type Description Exception

    Retrieval-specific errors defined by the implementation.

    "},{"location":"xlsx/client/#omniread.xlsx.client.FileSystemXlsxClient","title":"FileSystemXlsxClient","text":"

    Bases: BaseXlsxClient

    XLSX client that reads from the local filesystem.

    Notes

    Guarantees:

    - This client reads spreadsheet files directly from the disk and\n  returns their raw binary contents.\n
    "},{"location":"xlsx/client/#omniread.xlsx.client.FileSystemXlsxClient-functions","title":"Functions","text":""},{"location":"xlsx/client/#omniread.xlsx.client.FileSystemXlsxClient.fetch","title":"fetch","text":"
    fetch(path: Path) -> bytes\n

    Read an xlsx file from the local filesystem.

    Parameters:

    Name Type Description Default path Path

    Filesystem path to the spreadsheet file.

    required

    Returns:

    Name Type Description bytes bytes

    Raw xlsx bytes.

    Raises:

    Type Description FileNotFoundError

    If the path does not exist.

    ValueError

    If the path exists but is not a file.

    "},{"location":"xlsx/parser/","title":"Parser","text":""},{"location":"xlsx/parser/#omniread.xlsx.parser","title":"omniread.xlsx.parser","text":""},{"location":"xlsx/parser/#omniread.xlsx.parser--summary","title":"Summary","text":"

    XLSX parser implementations for OmniRead.

    This module provides a concrete, generic parser for Office Open XML spreadsheets. It exposes workbook sheets as lists of string rows so downstream consumers can interpret tabular content without depending on openpyxl directly.

    The parser is intentionally statement-agnostic: it performs no header detection or column interpretation beyond basic cell normalization.

    "},{"location":"xlsx/parser/#omniread.xlsx.parser-classes","title":"Classes","text":""},{"location":"xlsx/parser/#omniread.xlsx.parser.XlsxParser","title":"XlsxParser","text":"
    XlsxParser(\n    content: Content,\n    *,\n    data_only: bool = True,\n    read_only: bool = True\n)\n

    Bases: XlsxParserBase[list[list[str]]]

    Generic xlsx parser producing string rows from a worksheet.

    Notes

    Responsibilities:

    - Lazily load the workbook owned by the parser's content.\n- Normalize cells (including dates and numeric values) into\n  deterministic string representations.\n- Expose sheet discovery and row extraction helpers.\n

    Constraints:

    - Cells are rendered with ``str(value)`` after trimming; date and\n  datetime values are rendered in ISO format. Consumers requiring\n  locale-specific formatting must convert on their side.\n

    Initialize the parser.

    Parameters:

    Name Type Description Default content Content

    XLSX content to parse; its type must be supported.

    required data_only bool

    Passed to openpyxl: when True, formula cells yield their last computed value instead of the formula string.

    True read_only bool

    Passed to openpyxl: streaming mode for lower memory usage.

    True"},{"location":"xlsx/parser/#omniread.xlsx.parser.XlsxParser-attributes","title":"Attributes","text":""},{"location":"xlsx/parser/#omniread.xlsx.parser.XlsxParser.sheet_names","title":"sheet_names property","text":"
    sheet_names: list[str]\n

    Names of all worksheets contained in the workbook.

    "},{"location":"xlsx/parser/#omniread.xlsx.parser.XlsxParser.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {XLSX}\n

    Set of content types supported by this parser (XLSX only).

    "},{"location":"xlsx/parser/#omniread.xlsx.parser.XlsxParser.workbook","title":"workbook property","text":"
    workbook: Workbook\n

    The lazily loaded workbook backing this parser's content.

    "},{"location":"xlsx/parser/#omniread.xlsx.parser.XlsxParser-functions","title":"Functions","text":""},{"location":"xlsx/parser/#omniread.xlsx.parser.XlsxParser.parse","title":"parse","text":"
    parse() -> list[list[str]]\n

    Parse the first worksheet into normalized string rows.

    Returns:

    Type Description list[list[str]]

    list[list[str]]: Rows of the default (first) worksheet.

    "},{"location":"xlsx/parser/#omniread.xlsx.parser.XlsxParser.rows","title":"rows","text":"
    rows(\n    sheet: int | str | None = None,\n    *,\n    skip_empty: bool = True\n) -> list[list[str]]\n

    Extract normalized string rows from a worksheet.

    Parameters:

    Name Type Description Default sheet int | str | None

    Worksheet index or title; defaults to the first worksheet.

    None skip_empty bool

    When True (default), rows whose cells are all blank are omitted.

    True

    Returns:

    Type Description list[list[str]]

    list[list[str]]: Normalized rows; trailing blank cells are trimmed per row.

    Raises:

    Type Description ValueError

    If the requested sheet does not exist.

    "},{"location":"xlsx/parser/#omniread.xlsx.parser.XlsxParser.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"xlsx/parser_base/","title":"Parser Base","text":""},{"location":"xlsx/parser_base/#omniread.xlsx.parser_base","title":"omniread.xlsx.parser_base","text":""},{"location":"xlsx/parser_base/#omniread.xlsx.parser_base--summary","title":"Summary","text":"

    XLSX parser base implementation for OmniRead.

    This module defines the XLSX-specific parser contract, extending the format-agnostic BaseParser with constraints appropriate for Office Open XML spreadsheet content.

    "},{"location":"xlsx/parser_base/#omniread.xlsx.parser_base-classes","title":"Classes","text":""},{"location":"xlsx/parser_base/#omniread.xlsx.parser_base.XlsxParserBase","title":"XlsxParserBase","text":"
    XlsxParserBase(content: Content)\n

    Bases: BaseParser[T], Generic[T]

    Base xlsx parser.

    Notes

    Responsibilities:

    - This class enforces xlsx content-type compatibility and provides\n  the extension point for implementing concrete xlsx parsing\n  strategies.\n

    Constraints:

    - Concrete implementations must define the output type `T` and\n  implement the `parse()` method.\n

    Initialize the parser with content to be parsed.

    Parameters:

    Name Type Description Default content Content

    Content instance to be parsed.

    required

    Raises:

    Type Description ValueError

    If the content type is not supported by this parser.

    "},{"location":"xlsx/parser_base/#omniread.xlsx.parser_base.XlsxParserBase-attributes","title":"Attributes","text":""},{"location":"xlsx/parser_base/#omniread.xlsx.parser_base.XlsxParserBase.supported_types","title":"supported_types class-attribute instance-attribute","text":"
    supported_types: set[ContentType] = {XLSX}\n

    Set of content types supported by this parser (XLSX only).

    "},{"location":"xlsx/parser_base/#omniread.xlsx.parser_base.XlsxParserBase-functions","title":"Functions","text":""},{"location":"xlsx/parser_base/#omniread.xlsx.parser_base.XlsxParserBase.parse","title":"parse abstractmethod","text":"
    parse() -> T\n

    Parse xlsx content into a structured output.

    Returns:

    Name Type Description T T

    Parsed representation of type T.

    Raises:

    Type Description Exception

    Parsing-specific errors as defined by the implementation.

    "},{"location":"xlsx/parser_base/#omniread.xlsx.parser_base.XlsxParserBase.supports","title":"supports","text":"
    supports() -> bool\n

    Check whether this parser supports the content's type.

    Returns:

    Name Type Description bool bool

    True if the content type is supported; False otherwise.

    "},{"location":"xlsx/scraper/","title":"Scraper","text":""},{"location":"xlsx/scraper/#omniread.xlsx.scraper","title":"omniread.xlsx.scraper","text":""},{"location":"xlsx/scraper/#omniread.xlsx.scraper--summary","title":"Summary","text":"

    XLSX scraper for OmniRead.

    This module defines the scraper responsible for acquiring raw Office Open XML spreadsheet content from a backing store via a configured client.

    The scraper does not interpret or parse the acquired bytes; it wraps them in the canonical Content model.

    "},{"location":"xlsx/scraper/#omniread.xlsx.scraper-classes","title":"Classes","text":""},{"location":"xlsx/scraper/#omniread.xlsx.scraper.XlsxScraper","title":"XlsxScraper","text":"
    XlsxScraper(*, client: BaseXlsxClient)\n

    Scraper for xlsx spreadsheet documents.

    Notes

    Responsibilities:

    - Fetch raw xlsx bytes via the configured client.\n- Wrap the payload in a canonical `Content` instance with the\n  XLSX content type and source identifier.\n

    Constraints:

    - The scraper does not perform parsing or interpretation.\n- Does not assume a specific storage backend.\n

    Initialize the XLSX scraper.

    Parameters:

    Name Type Description Default client BaseXlsxClient

    Client responsible for retrieving raw spreadsheet bytes.

    required"},{"location":"xlsx/scraper/#omniread.xlsx.scraper.XlsxScraper-functions","title":"Functions","text":""},{"location":"xlsx/scraper/#omniread.xlsx.scraper.XlsxScraper.fetch","title":"fetch","text":"
    fetch(\n    source: Any,\n    *,\n    metadata: Mapping[str, Any] | None = None\n) -> Content\n

    Fetch an xlsx document from the given source.

    Parameters:

    Name Type Description Default source Any

    Identifier of the spreadsheet source as understood by the configured client.

    required metadata Mapping[str, Any] | None

    Optional metadata to attach to the returned content.

    None

    Returns:

    Name Type Description Content Content

    A Content instance containing raw xlsx bytes, source identifier, XLSX content type, and optional metadata.

    Raises:

    Type Description Exception

    Retrieval-specific errors raised by the client.

    "}]} \ No newline at end of file diff --git a/omniread/lib/sitemap.xml.gz b/omniread/lib/sitemap.xml.gz index 01465253604ef76b3b2b517f1166919160e58317..89851db59432aa6d7a5d0fe339808593e07e45ef 100644 GIT binary patch delta 13 Ucmb=gXP58h;9$73aw2;L033S+RR910 delta 13 Ucmb=gXP58h;ArStF_FCj03F=~f&c&j diff --git a/omniread/lib/xlsx/client/index.html b/omniread/lib/xlsx/client/index.html index e33caaf..979f2d5 100644 --- a/omniread/lib/xlsx/client/index.html +++ b/omniread/lib/xlsx/client/index.html @@ -9,6 +9,10 @@ + + + + @@ -639,6 +643,499 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + + + +
  • + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + +
  • + + +
    diff --git a/omniread/lib/xlsx/index.html b/omniread/lib/xlsx/index.html index 7c8ed73..45c9ea1 100644 --- a/omniread/lib/xlsx/index.html +++ b/omniread/lib/xlsx/index.html @@ -9,6 +9,10 @@ + + + + @@ -639,6 +643,340 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + + + +
  • + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + +
  • + + + @@ -1416,7 +1754,7 @@ object storage key, or remote reference.

    content - Content + Content
    @@ -1766,7 +2104,7 @@ Normalized rows; trailing blank cells are trimmed per row.

    - Bases: BaseParser[T], Generic[T]

    + Bases: BaseParser[T], Generic[T]

    Base xlsx parser.

    @@ -1804,7 +2142,7 @@ Normalized rows; trailing blank cells are trimmed per row.

    content - Content + Content
    @@ -2153,7 +2491,7 @@ configured client.

    Content - Content + Content
    diff --git a/omniread/lib/xlsx/parser/index.html b/omniread/lib/xlsx/parser/index.html index c35fe06..bdb4c0c 100644 --- a/omniread/lib/xlsx/parser/index.html +++ b/omniread/lib/xlsx/parser/index.html @@ -9,6 +9,10 @@ + + + + @@ -639,6 +643,520 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + + + +
  • + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + +
  • + + +
    @@ -920,7 +1438,7 @@ detection or column interpretation beyond basic cell normalization.

    content - Content + Content
    diff --git a/omniread/lib/xlsx/parser_base/index.html b/omniread/lib/xlsx/parser_base/index.html index 5efed66..314a2f5 100644 --- a/omniread/lib/xlsx/parser_base/index.html +++ b/omniread/lib/xlsx/parser_base/index.html @@ -9,6 +9,10 @@ + + + + @@ -639,6 +643,493 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + + + +
  • + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + +
  • + + +
    @@ -838,7 +1329,7 @@ XML spreadsheet content.

    - Bases: BaseParser[T], Generic[T]

    + Bases: BaseParser[T], Generic[T]

    Base xlsx parser.

    @@ -876,7 +1367,7 @@ XML spreadsheet content.

    content - Content + Content
    diff --git a/omniread/lib/xlsx/scraper/index.html b/omniread/lib/xlsx/scraper/index.html index cca5b7d..8aa3f09 100644 --- a/omniread/lib/xlsx/scraper/index.html +++ b/omniread/lib/xlsx/scraper/index.html @@ -9,6 +9,8 @@ + + @@ -639,6 +641,460 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + + + +
  • + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
  • + + + + + + + + + + +
  • + + +
    @@ -948,7 +1404,7 @@ configured client.

    Content - Content + Content
    diff --git a/omniread/wiki/01_overview/index.html b/omniread/wiki/01_overview/index.html new file mode 100644 index 0000000..03afe8c --- /dev/null +++ b/omniread/wiki/01_overview/index.html @@ -0,0 +1,751 @@ + + + + + + + + + + + + + + + + + + + + + + + Overview - OmniRead Documentation + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
    + +
    + + + + +
    + + +
    + +
    + + + + + + + + + +
    +
    + + + +
    +
    +
    + + + + + + + + + +
    +
    +
    + + + + +
    +
    + + + + + +

    🧱 Overview

    +

    OmniRead is designed as a decoupled content engine with three distinct +layers. Understanding them is the key to using and extending the library.

    +
    +

    🏗️ Architecture

    +
     1
    + 2
    + 3
    + 4
    + 5
    + 6
    + 7
    + 8
    + 9
    +10
    +11
    +12
    +13
    +14
    +15
                     ┌─────────────────────────────┐
    +                 │  Source (URL, file, storage) │
    +                 └────────────┬────────────────┘
    +                              │
    +                    ┌─────────▼──────────┐
    +                    │   Scraper / Client  │  fetches raw bytes
    +                    └─────────┬──────────┘
    +                              │  returns
    +                    ┌─────────▼──────────┐
    +                    │     Content         │  raw + source + type
    +                    └─────────┬──────────┘
    +                              │
    +                    ┌─────────▼──────────┐
    +                    │      Parser         │  parse() → structured T
    +                    └────────────────────┘
    +
    +
      +
    1. Scraper (BaseScraper) — fetches raw bytes from a source (HTTP URL, + filesystem path, object storage). Returns a Content instance. + Scrapers never interpret content.
    2. +
    3. Content (Content) — the canonical exchange model: raw bytes, + source identifier, and optional ContentType enum.
    4. +
    5. Parser (BaseParser[T]) — receives a Content and returns a + structured result of type T via parse().
    6. +
    +
    +

    📦 The Content model

    +

    Defined in omniread.core.content:

    +
    1
    +2
    +3
    +4
    +5
    +6
    +7
    +8
    from dataclasses import dataclass
    +from omniread import Content, ContentType
    +
    +@dataclass(slots=True)
    +class Content:
    +    raw: bytes
    +    source: str
    +    content_type: ContentType | None = None
    +
    +
      +
    • raw — the raw bytes exactly as retrieved.
    • +
    • source — URL, file path, or logical name identifying the origin.
    • +
    • content_type — optional ContentType enum value.
    • +
    +
    +

    🎭 The ContentType enum

    + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
    ValueMIMEUsed by
    HTMLtext/htmlHTMLScraper
    PDFapplication/pdfPDFScraper
    XLSXapplication/vnd.openxmlformats-...XlsxScraper
    CSVtext/csvCsvScraper
    JSONapplication/json
    XMLapplication/xml
    +
    +

    🧩 Format modules at a glance

    + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
    ModuleScraperParserClientNotes
    omniread.htmlHTMLScraperHTMLParserhttpx + BeautifulSoup
    omniread.pdfPDFScraperPDFParserFileSystemPDFClientexplicit client layer
    omniread.csvCsvScraperCsvParserFileSystemCsvClientstdlib csv module
    omniread.xlsxXlsxScraperXlsxParserFileSystemXlsxClientopenpyxl-backed
    +
    + + + + + + + + + + + + + + + +
    +
    + + + + + +
    + + + +
    + + + +
    +
    +
    +
    + + + + + + + + + + + + \ No newline at end of file diff --git a/omniread/wiki/02_how_to_use/index.html b/omniread/wiki/02_how_to_use/index.html new file mode 100644 index 0000000..43782be --- /dev/null +++ b/omniread/wiki/02_how_to_use/index.html @@ -0,0 +1,724 @@ + + + + + + + + + + + + + + + + + + + + + + + How to Use - OmniRead Documentation + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
    + +
    + + + + +
    + + +
    + +
    + + + + + + + + + +
    +
    + + + +
    +
    +
    + + + + + + + + + +
    +
    +
    + + + + +
    +
    + + + + + +

    🖥️ How to Use

    +

    Every format follows the same pipeline: scrape → Content → parse. +This page shows each supported format with working patterns.

    +
    +

    🌐 HTML

    +
     1
    + 2
    + 3
    + 4
    + 5
    + 6
    + 7
    + 8
    + 9
    +10
    from omniread import HTMLScraper, HTMLParser
    +
    +class TitleParser(HTMLParser[str]):
    +    def parse(self) -> str:
    +        return self._soup.title.string
    +
    +scraper = HTMLScraper()                     # httpx under the hood
    +content = scraper.fetch("https://example.com")
    +
    +title = TitleParser(content).parse()
    +
    +

    HTMLScraper accepts an optional client (an httpx.Client) for transport +control — the test suite wires one to a mock transport.

    +
    +

    📕 PDF

    +

    PDFs need a client to supply raw bytes before parsing:

    +
     1
    + 2
    + 3
    + 4
    + 5
    + 6
    + 7
    + 8
    + 9
    +10
    +11
    +12
    +13
    from pathlib import Path
    +from omniread import FileSystemPDFClient, PDFScraper, PDFParser
    +
    +class TextPDFParser(PDFParser[str]):
    +    def parse(self) -> str:
    +        # implement your extraction logic
    +        return self.content.raw  # bytes, decode as needed
    +
    +client = FileSystemPDFClient()
    +scraper = PDFScraper(client=client)
    +content = scraper.fetch(Path("document.pdf"))
    +
    +result = TextPDFParser(content).parse()
    +
    +

    PDFParser subclasses receive self.content and implement parse().

    +
    +

    📊 CSV

    +
    1
    +2
    +3
    +4
    +5
    +6
    +7
    +8
    from omniread import FileSystemCsvClient, CsvScraper, CsvParser
    +
    +scraper = CsvScraper(client=FileSystemCsvClient())
    +content = scraper.fetch("data.csv")
    +
    +parser = CsvParser(content)
    +for row in parser.rows():
    +    print(row)
    +
    +

    CsvParser.rows() yields string rows trimmed of empties by default +(skip_empty=True).

    +
    +

    📑 XLSX

    +
    1
    +2
    +3
    +4
    +5
    +6
    +7
    +8
    +9
    from omniread import FileSystemXlsxClient, XlsxScraper, XlsxParser
    +
    +scraper = XlsxScraper(client=FileSystemXlsxClient())
    +content = scraper.fetch("statement.xlsx")
    +
    +parser = XlsxParser(content)
    +print(parser.sheet_names)                 # e.g. ["Statement"]
    +rows = parser.rows(sheet="Statement")     # by name or index
    +all_rows = parser.parse()                 # alias for rows()
    +
    +

    Key behaviors:

    +
      +
    • rows(skip_empty=False) keeps blank rows (off by default).
    • +
    • Cells render as trimmed strings; date cells convert to ISO format + (2026-06-01T00:00:00).
    • +
    • rows(sheet="Missing") raises for an unknown sheet.
    • +
    +
    +

    🔀 End-to-end flow

    +
    1
    +2
    +3
    +4
    +5
    +6
    content = scraper.fetch(source)   # 1. acquire → Content
    +assert isinstance(content.raw, bytes)
    +assert content.content_type is not None
    +
    +parser = MyParser(content)        # 2. interpret → T
    +result = parser.parse()
    +
    +
    + + + + + + + + + + + + + + + +
    +
    + + + + + +
    + + + +
    + + + +
    +
    +
    +
    + + + + + + + + + + + + \ No newline at end of file diff --git a/omniread/wiki/03_extending/index.html b/omniread/wiki/03_extending/index.html new file mode 100644 index 0000000..233eddb --- /dev/null +++ b/omniread/wiki/03_extending/index.html @@ -0,0 +1,714 @@ + + + + + + + + + + + + + + + + + + + + + + + Extending OmniRead - OmniRead Documentation + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
    + +
    + + + + +
    + + +
    + +
    + + + + + + + + + +
    +
    + + + +
    +
    +
    + + + + + + + + + +
    +
    +
    + + + + +
    +
    + + + + + +

    🧩 Extending OmniRead

    +

    OmniRead is meant to be extended by subclassing. All public extension points +are generic over their result type, so your parser returns exactly the shape +you need.

    +
    +

    🧬 Custom parsers

    +

    Subclass BaseParser[T] (or a format parser) and implement parse():

    +
     1
    + 2
    + 3
    + 4
    + 5
    + 6
    + 7
    + 8
    + 9
    +10
    +11
    +12
    +13
    +14
    +15
    from pydantic import BaseModel
    +from omniread import HTMLParser
    +
    +class Page(BaseModel):
    +    title: str
    +    content: str | None
    +
    +class PageParser(HTMLParser[Page]):
    +    def parse(self) -> Page:
    +        soup = self._soup
    +        div = soup.find("div", id="content")
    +        return Page(
    +            title=soup.title.string,
    +            content=div.get_text() if div else None,
    +        )
    +
    +

    The parsed page is validated by Pydantic on construction — no manual +assertions required.

    +
    +

    🧬 Custom PDF parsers

    +

    PDF binary layout is format-specific, so parsers return your own model:

    +
     1
    + 2
    + 3
    + 4
    + 5
    + 6
    + 7
    + 8
    + 9
    +10
    +11
    +12
    +13
    from typing import Literal
    +from pydantic import BaseModel
    +from omniread import PDFParser
    +
    +class ParsedPDF(BaseModel):
    +    size_bytes: int
    +    magic: Literal[b"%PDF"]
    +
    +class SimplePDFParser(PDFParser[ParsedPDF]):
    +    def parse(self) -> ParsedPDF:
    +        if not self.content.raw.startswith(b"%PDF"):
    +            raise ValueError("Not a valid PDF")
    +        return ParsedPDF(size_bytes=len(self.content.raw), magic=b"%PDF")
    +
    +
    +

    🧬 Custom clients

    +

    Clients supply raw bytes to a scraper. For PDFs, subclass +BasePDFClient (or FileSystemPDFClient) and implement +fetch(source) -> bytes:

    +
    1
    +2
    +3
    +4
    +5
    from omniread.pdf.client import BasePDFClient
    +
    +class MockPDFClient(BasePDFClient):
    +    def fetch(self, source):
    +        return b"%PDF ..."  # bytes for the logical identifier
    +
    +

    The same pattern applies to BaseCsvClient and BaseXlsxClient.

    +
    +

    🚀 Custom scrapers

    +

    Festch something that a built-in scraper does not cover by extending +BaseScraper:

    +
    1
    +2
    +3
    +4
    +5
    +6
    from omniread import BaseScraper, Content, ContentType
    +
    +class StorageScraper(BaseScraper):
    +    def fetch(self, source, *, metadata=None):
    +        raw = my_object_storage.download(source)      # your I/O
    +        return Content(raw=raw, source=source, content_type=ContentType.JSON)
    +
    +
    +

    ✅ Extension checklist

    +
      +
    1. Keep scraper and parser separate — never mix I/O into parse().
    2. +
    3. Return Content from any scraper/client so downstream stays uniform.
    4. +
    5. Return a typed result from your parser (Pydantic model, dataclass, str).
    6. +
    7. Test your custom layers with a mock client, not a live network.
    8. +
    +
    + + + + + + + + + + + + + + + +
    +
    + + + + + +
    + + + +
    + + + +
    +
    +
    +
    + + + + + + + + + + + + \ No newline at end of file diff --git a/omniread/wiki/04_development/index.html b/omniread/wiki/04_development/index.html new file mode 100644 index 0000000..47b8bed --- /dev/null +++ b/omniread/wiki/04_development/index.html @@ -0,0 +1,696 @@ + + + + + + + + + + + + + + + + + + + + + Development - OmniRead Documentation + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
    + +
    + + + + +
    + + +
    + +
    + + + + + + + + + +
    +
    + + + +
    +
    +
    + + + + + + + + + +
    +
    +
    + + + + +
    +
    + + + + + +

    🛠️ Development

    +

    Working on omniread itself.

    +
    +

    📂 Repository layout

    + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
    PathPurpose
    omniread/The library package (core + html, pdf, csv, xlsx modules)
    omniread/*.pyiType stubs kept in sync with implementations
    tests/End-to-end and unit tests against mock transports/clients
    covers
    docs/lib/Generated library reference (docforge, flat layout)
    docs/mcp/Machine-readable bundle served by the MCP server
    docs/wiki/This hand-written wiki
    +
    +

    🔧 Setup

    +
    1
    +2
    python -m venv .venv
    +.venv/Scripts/pip install -e ".[dev]"
    +
    +
    +

    🧪 Tests

    +

    Run the suite (offline; a mock httpx transport and mock PDF client are used):

    +
    .venv/Scripts/pytest
    +
    +

    Coverage spans end-to-end scrape → parse flows for HTML, PDF, and XLSX, +plus client validation and CSV/XLSX parsing edge cases.

    +
    +

    ✅ Quality gates

    +

    The CI quality gate runs, matching the Drone pipeline:

    +
    1
    +2
    +3
    +4
    .venv/Scripts/black --check .
    +.venv/Scripts/ruff check .
    +.venv/Scripts/mypy
    +.venv/Scripts/pytest
    +
    +
    +

    📝 Building documentation (docforge)

    +

    The site is generated by docforge +and served per kind under site/{kind}:

    +
    1
    +2
    +3
    +4
    doc-forge build \
    +  --mkdocs --mcp \
    +  --module-is-source --module omniread \
    +  --site-name "OmniRead"
    +
    +
      +
    • --module-is-source renders the flat docs/lib/ layout (no nesting under + omniread/), matching docforge.nav.yml and docs/mkdocs.lib.yml.
    • +
    • --mcp regenerates the structured bundle in docs/mcp/.
    • +
    • --wiki builds this wiki.
    • +
    • gen_api.py (if present) regenerates the API docs.
    • +
    +

    Preview locally:

    +
    1
    +2
    +3
    doc-forge serve --lib
    +doc-forge serve --wiki
    +doc-forge serve --mcp
    +
    +
    + + + + + + + + + + + + + + + +
    +
    + + + + + +
    + + + +
    + + + +
    +
    +
    +
    + + + + + + + + + + + + \ No newline at end of file diff --git a/omniread/wiki/404.html b/omniread/wiki/404.html new file mode 100644 index 0000000..ade9541 --- /dev/null +++ b/omniread/wiki/404.html @@ -0,0 +1,478 @@ + + + + + + + + + + + + + + + + + + + OmniRead Documentation + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
    + +
    +
    + +
    + + + + +
    + + +
    + +
    + + + + + + + + + +
    +
    + + + +
    +
    +
    + + + + + + + + + +
    +
    +
    + + + + +
    +
    + +

    404 - Not found

    + +
    +
    + + + + + +
    + + + +
    + + + +
    +
    +
    +
    + + + + + + + + + + + + \ No newline at end of file diff --git a/omniread/wiki/assets/images/favicon.png b/omniread/wiki/assets/images/favicon.png new file mode 100644 index 0000000000000000000000000000000000000000..1cf13b9f9d978896599290a74f77d5dbe7d1655c GIT binary patch literal 1870 zcmV-U2eJ5xP)Gc)JR9QMau)O=X#!i9;T z37kk-upj^(fsR36MHs_+1RCI)NNu9}lD0S{B^g8PN?Ww(5|~L#Ng*g{WsqleV}|#l zz8@ri&cTzw_h33bHI+12+kK6WN$h#n5cD8OQt`5kw6p~9H3()bUQ8OS4Q4HTQ=1Ol z_JAocz`fLbT2^{`8n~UAo=#AUOf=SOq4pYkt;XbC&f#7lb$*7=$na!mWCQ`dBQsO0 zLFBSPj*N?#u5&pf2t4XjEGH|=pPQ8xh7tpx;US5Cx_Ju;!O`ya-yF`)b%TEt5>eP1ZX~}sjjA%FJF?h7cX8=b!DZl<6%Cv z*G0uvvU+vmnpLZ2paivG-(cd*y3$hCIcsZcYOGh{$&)A6*XX&kXZd3G8m)G$Zz-LV z^GF3VAW^Mdv!)4OM8EgqRiz~*Cji;uzl2uC9^=8I84vNp;ltJ|q-*uQwGp2ma6cY7 z;`%`!9UXO@fr&Ebapfs34OmS9^u6$)bJxrucutf>`dKPKT%%*d3XlFVKunp9 zasduxjrjs>f8V=D|J=XNZp;_Zy^WgQ$9WDjgY=z@stwiEBm9u5*|34&1Na8BMjjgf3+SHcr`5~>oz1Y?SW^=K z^bTyO6>Gar#P_W2gEMwq)ot3; zREHn~U&Dp0l6YT0&k-wLwYjb?5zGK`W6S2v+K>AM(95m2C20L|3m~rN8dprPr@t)5lsk9Hu*W z?pS990s;Ez=+Rj{x7p``4>+c0G5^pYnB1^!TL=(?HLHZ+HicG{~4F1d^5Awl_2!1jICM-!9eoLhbbT^;yHcefyTAaqRcY zmuctDopPT!%k+}x%lZRKnzykr2}}XfG_ne?nRQO~?%hkzo;@RN{P6o`&mMUWBYMTe z6i8ChtjX&gXl`nvrU>jah)2iNM%JdjqoaeaU%yVn!^70x-flljp6Q5tK}5}&X8&&G zX3fpb3E(!rH=zVI_9Gjl45w@{(ITqngWFe7@9{mX;tO25Z_8 zQHEpI+FkTU#4xu>RkN>b3Tnc3UpWzPXWm#o55GKF09j^Mh~)K7{QqbO_~(@CVq! zS<8954|P8mXN2MRs86xZ&Q4EfM@JB94b=(YGuk)s&^jiSF=t3*oNK3`rD{H`yQ?d; ztE=laAUoZx5?RC8*WKOj`%LXEkgDd>&^Q4M^z`%u0rg-It=hLCVsq!Z%^6eB-OvOT zFZ28TN&cRmgU}Elrnk43)!>Z1FCPL2K$7}gwzIc48NX}#!A1BpJP?#v5wkNprhV** z?Cpalt1oH&{r!o3eSKc&ap)iz2BTn_VV`4>9M^b3;(YY}4>#ML6{~(4mH+?%07*qo IM6N<$f(jP3KmY&$ literal 0 HcmV?d00001 diff --git a/omniread/wiki/assets/javascripts/bundle.f55a23d4.min.js b/omniread/wiki/assets/javascripts/bundle.f55a23d4.min.js new file mode 100644 index 0000000..01a46ad --- /dev/null +++ b/omniread/wiki/assets/javascripts/bundle.f55a23d4.min.js @@ -0,0 +1,16 @@ +"use strict";(()=>{var Wi=Object.create;var gr=Object.defineProperty;var Vi=Object.getOwnPropertyDescriptor;var Di=Object.getOwnPropertyNames,Vt=Object.getOwnPropertySymbols,zi=Object.getPrototypeOf,yr=Object.prototype.hasOwnProperty,ao=Object.prototype.propertyIsEnumerable;var io=(e,t,r)=>t in e?gr(e,t,{enumerable:!0,configurable:!0,writable:!0,value:r}):e[t]=r,$=(e,t)=>{for(var r in t||(t={}))yr.call(t,r)&&io(e,r,t[r]);if(Vt)for(var r of Vt(t))ao.call(t,r)&&io(e,r,t[r]);return e};var so=(e,t)=>{var r={};for(var o in e)yr.call(e,o)&&t.indexOf(o)<0&&(r[o]=e[o]);if(e!=null&&Vt)for(var o of Vt(e))t.indexOf(o)<0&&ao.call(e,o)&&(r[o]=e[o]);return r};var xr=(e,t)=>()=>(t||e((t={exports:{}}).exports,t),t.exports);var Ni=(e,t,r,o)=>{if(t&&typeof t=="object"||typeof t=="function")for(let n of Di(t))!yr.call(e,n)&&n!==r&&gr(e,n,{get:()=>t[n],enumerable:!(o=Vi(t,n))||o.enumerable});return e};var Lt=(e,t,r)=>(r=e!=null?Wi(zi(e)):{},Ni(t||!e||!e.__esModule?gr(r,"default",{value:e,enumerable:!0}):r,e));var co=(e,t,r)=>new Promise((o,n)=>{var i=p=>{try{s(r.next(p))}catch(c){n(c)}},a=p=>{try{s(r.throw(p))}catch(c){n(c)}},s=p=>p.done?o(p.value):Promise.resolve(p.value).then(i,a);s((r=r.apply(e,t)).next())});var lo=xr((Er,po)=>{(function(e,t){typeof Er=="object"&&typeof po!="undefined"?t():typeof define=="function"&&define.amd?define(t):t()})(Er,(function(){"use strict";function e(r){var o=!0,n=!1,i=null,a={text:!0,search:!0,url:!0,tel:!0,email:!0,password:!0,number:!0,date:!0,month:!0,week:!0,time:!0,datetime:!0,"datetime-local":!0};function s(k){return!!(k&&k!==document&&k.nodeName!=="HTML"&&k.nodeName!=="BODY"&&"classList"in k&&"contains"in k.classList)}function p(k){var ft=k.type,qe=k.tagName;return!!(qe==="INPUT"&&a[ft]&&!k.readOnly||qe==="TEXTAREA"&&!k.readOnly||k.isContentEditable)}function c(k){k.classList.contains("focus-visible")||(k.classList.add("focus-visible"),k.setAttribute("data-focus-visible-added",""))}function l(k){k.hasAttribute("data-focus-visible-added")&&(k.classList.remove("focus-visible"),k.removeAttribute("data-focus-visible-added"))}function f(k){k.metaKey||k.altKey||k.ctrlKey||(s(r.activeElement)&&c(r.activeElement),o=!0)}function u(k){o=!1}function d(k){s(k.target)&&(o||p(k.target))&&c(k.target)}function y(k){s(k.target)&&(k.target.classList.contains("focus-visible")||k.target.hasAttribute("data-focus-visible-added"))&&(n=!0,window.clearTimeout(i),i=window.setTimeout(function(){n=!1},100),l(k.target))}function L(k){document.visibilityState==="hidden"&&(n&&(o=!0),X())}function X(){document.addEventListener("mousemove",J),document.addEventListener("mousedown",J),document.addEventListener("mouseup",J),document.addEventListener("pointermove",J),document.addEventListener("pointerdown",J),document.addEventListener("pointerup",J),document.addEventListener("touchmove",J),document.addEventListener("touchstart",J),document.addEventListener("touchend",J)}function ee(){document.removeEventListener("mousemove",J),document.removeEventListener("mousedown",J),document.removeEventListener("mouseup",J),document.removeEventListener("pointermove",J),document.removeEventListener("pointerdown",J),document.removeEventListener("pointerup",J),document.removeEventListener("touchmove",J),document.removeEventListener("touchstart",J),document.removeEventListener("touchend",J)}function J(k){k.target.nodeName&&k.target.nodeName.toLowerCase()==="html"||(o=!1,ee())}document.addEventListener("keydown",f,!0),document.addEventListener("mousedown",u,!0),document.addEventListener("pointerdown",u,!0),document.addEventListener("touchstart",u,!0),document.addEventListener("visibilitychange",L,!0),X(),r.addEventListener("focus",d,!0),r.addEventListener("blur",y,!0),r.nodeType===Node.DOCUMENT_FRAGMENT_NODE&&r.host?r.host.setAttribute("data-js-focus-visible",""):r.nodeType===Node.DOCUMENT_NODE&&(document.documentElement.classList.add("js-focus-visible"),document.documentElement.setAttribute("data-js-focus-visible",""))}if(typeof window!="undefined"&&typeof document!="undefined"){window.applyFocusVisiblePolyfill=e;var t;try{t=new CustomEvent("focus-visible-polyfill-ready")}catch(r){t=document.createEvent("CustomEvent"),t.initCustomEvent("focus-visible-polyfill-ready",!1,!1,{})}window.dispatchEvent(t)}typeof document!="undefined"&&e(document)}))});var qr=xr((dy,On)=>{"use strict";/*! + * escape-html + * Copyright(c) 2012-2013 TJ Holowaychuk + * Copyright(c) 2015 Andreas Lubbe + * Copyright(c) 2015 Tiancheng "Timothy" Gu + * MIT Licensed + */var $a=/["'&<>]/;On.exports=Pa;function Pa(e){var t=""+e,r=$a.exec(t);if(!r)return t;var o,n="",i=0,a=0;for(i=r.index;i{/*! + * clipboard.js v2.0.11 + * https://clipboardjs.com/ + * + * Licensed MIT © Zeno Rocha + */(function(t,r){typeof Rt=="object"&&typeof Yr=="object"?Yr.exports=r():typeof define=="function"&&define.amd?define([],r):typeof Rt=="object"?Rt.ClipboardJS=r():t.ClipboardJS=r()})(Rt,function(){return(function(){var e={686:(function(o,n,i){"use strict";i.d(n,{default:function(){return Ui}});var a=i(279),s=i.n(a),p=i(370),c=i.n(p),l=i(817),f=i.n(l);function u(D){try{return document.execCommand(D)}catch(A){return!1}}var d=function(A){var M=f()(A);return u("cut"),M},y=d;function L(D){var A=document.documentElement.getAttribute("dir")==="rtl",M=document.createElement("textarea");M.style.fontSize="12pt",M.style.border="0",M.style.padding="0",M.style.margin="0",M.style.position="absolute",M.style[A?"right":"left"]="-9999px";var F=window.pageYOffset||document.documentElement.scrollTop;return M.style.top="".concat(F,"px"),M.setAttribute("readonly",""),M.value=D,M}var X=function(A,M){var F=L(A);M.container.appendChild(F);var V=f()(F);return u("copy"),F.remove(),V},ee=function(A){var M=arguments.length>1&&arguments[1]!==void 0?arguments[1]:{container:document.body},F="";return typeof A=="string"?F=X(A,M):A instanceof HTMLInputElement&&!["text","search","url","tel","password"].includes(A==null?void 0:A.type)?F=X(A.value,M):(F=f()(A),u("copy")),F},J=ee;function k(D){"@babel/helpers - typeof";return typeof Symbol=="function"&&typeof Symbol.iterator=="symbol"?k=function(M){return typeof M}:k=function(M){return M&&typeof Symbol=="function"&&M.constructor===Symbol&&M!==Symbol.prototype?"symbol":typeof M},k(D)}var ft=function(){var A=arguments.length>0&&arguments[0]!==void 0?arguments[0]:{},M=A.action,F=M===void 0?"copy":M,V=A.container,Y=A.target,$e=A.text;if(F!=="copy"&&F!=="cut")throw new Error('Invalid "action" value, use either "copy" or "cut"');if(Y!==void 0)if(Y&&k(Y)==="object"&&Y.nodeType===1){if(F==="copy"&&Y.hasAttribute("disabled"))throw new Error('Invalid "target" attribute. Please use "readonly" instead of "disabled" attribute');if(F==="cut"&&(Y.hasAttribute("readonly")||Y.hasAttribute("disabled")))throw new Error(`Invalid "target" attribute. You can't cut text from elements with "readonly" or "disabled" attributes`)}else throw new Error('Invalid "target" value, use a valid Element');if($e)return J($e,{container:V});if(Y)return F==="cut"?y(Y):J(Y,{container:V})},qe=ft;function Fe(D){"@babel/helpers - typeof";return typeof Symbol=="function"&&typeof Symbol.iterator=="symbol"?Fe=function(M){return typeof M}:Fe=function(M){return M&&typeof Symbol=="function"&&M.constructor===Symbol&&M!==Symbol.prototype?"symbol":typeof M},Fe(D)}function ki(D,A){if(!(D instanceof A))throw new TypeError("Cannot call a class as a function")}function no(D,A){for(var M=0;M0&&arguments[0]!==void 0?arguments[0]:{};this.action=typeof V.action=="function"?V.action:this.defaultAction,this.target=typeof V.target=="function"?V.target:this.defaultTarget,this.text=typeof V.text=="function"?V.text:this.defaultText,this.container=Fe(V.container)==="object"?V.container:document.body}},{key:"listenClick",value:function(V){var Y=this;this.listener=c()(V,"click",function($e){return Y.onClick($e)})}},{key:"onClick",value:function(V){var Y=V.delegateTarget||V.currentTarget,$e=this.action(Y)||"copy",Wt=qe({action:$e,container:this.container,target:this.target(Y),text:this.text(Y)});this.emit(Wt?"success":"error",{action:$e,text:Wt,trigger:Y,clearSelection:function(){Y&&Y.focus(),window.getSelection().removeAllRanges()}})}},{key:"defaultAction",value:function(V){return vr("action",V)}},{key:"defaultTarget",value:function(V){var Y=vr("target",V);if(Y)return document.querySelector(Y)}},{key:"defaultText",value:function(V){return vr("text",V)}},{key:"destroy",value:function(){this.listener.destroy()}}],[{key:"copy",value:function(V){var Y=arguments.length>1&&arguments[1]!==void 0?arguments[1]:{container:document.body};return J(V,Y)}},{key:"cut",value:function(V){return y(V)}},{key:"isSupported",value:function(){var V=arguments.length>0&&arguments[0]!==void 0?arguments[0]:["copy","cut"],Y=typeof V=="string"?[V]:V,$e=!!document.queryCommandSupported;return Y.forEach(function(Wt){$e=$e&&!!document.queryCommandSupported(Wt)}),$e}}]),M})(s()),Ui=Fi}),828:(function(o){var n=9;if(typeof Element!="undefined"&&!Element.prototype.matches){var i=Element.prototype;i.matches=i.matchesSelector||i.mozMatchesSelector||i.msMatchesSelector||i.oMatchesSelector||i.webkitMatchesSelector}function a(s,p){for(;s&&s.nodeType!==n;){if(typeof s.matches=="function"&&s.matches(p))return s;s=s.parentNode}}o.exports=a}),438:(function(o,n,i){var a=i(828);function s(l,f,u,d,y){var L=c.apply(this,arguments);return l.addEventListener(u,L,y),{destroy:function(){l.removeEventListener(u,L,y)}}}function p(l,f,u,d,y){return typeof l.addEventListener=="function"?s.apply(null,arguments):typeof u=="function"?s.bind(null,document).apply(null,arguments):(typeof l=="string"&&(l=document.querySelectorAll(l)),Array.prototype.map.call(l,function(L){return s(L,f,u,d,y)}))}function c(l,f,u,d){return function(y){y.delegateTarget=a(y.target,f),y.delegateTarget&&d.call(l,y)}}o.exports=p}),879:(function(o,n){n.node=function(i){return i!==void 0&&i instanceof HTMLElement&&i.nodeType===1},n.nodeList=function(i){var a=Object.prototype.toString.call(i);return i!==void 0&&(a==="[object NodeList]"||a==="[object HTMLCollection]")&&"length"in i&&(i.length===0||n.node(i[0]))},n.string=function(i){return typeof i=="string"||i instanceof String},n.fn=function(i){var a=Object.prototype.toString.call(i);return a==="[object Function]"}}),370:(function(o,n,i){var a=i(879),s=i(438);function p(u,d,y){if(!u&&!d&&!y)throw new Error("Missing required arguments");if(!a.string(d))throw new TypeError("Second argument must be a String");if(!a.fn(y))throw new TypeError("Third argument must be a Function");if(a.node(u))return c(u,d,y);if(a.nodeList(u))return l(u,d,y);if(a.string(u))return f(u,d,y);throw new TypeError("First argument must be a String, HTMLElement, HTMLCollection, or NodeList")}function c(u,d,y){return u.addEventListener(d,y),{destroy:function(){u.removeEventListener(d,y)}}}function l(u,d,y){return Array.prototype.forEach.call(u,function(L){L.addEventListener(d,y)}),{destroy:function(){Array.prototype.forEach.call(u,function(L){L.removeEventListener(d,y)})}}}function f(u,d,y){return s(document.body,u,d,y)}o.exports=p}),817:(function(o){function n(i){var a;if(i.nodeName==="SELECT")i.focus(),a=i.value;else if(i.nodeName==="INPUT"||i.nodeName==="TEXTAREA"){var s=i.hasAttribute("readonly");s||i.setAttribute("readonly",""),i.select(),i.setSelectionRange(0,i.value.length),s||i.removeAttribute("readonly"),a=i.value}else{i.hasAttribute("contenteditable")&&i.focus();var p=window.getSelection(),c=document.createRange();c.selectNodeContents(i),p.removeAllRanges(),p.addRange(c),a=p.toString()}return a}o.exports=n}),279:(function(o){function n(){}n.prototype={on:function(i,a,s){var p=this.e||(this.e={});return(p[i]||(p[i]=[])).push({fn:a,ctx:s}),this},once:function(i,a,s){var p=this;function c(){p.off(i,c),a.apply(s,arguments)}return c._=a,this.on(i,c,s)},emit:function(i){var a=[].slice.call(arguments,1),s=((this.e||(this.e={}))[i]||[]).slice(),p=0,c=s.length;for(p;p0&&i[i.length-1])&&(c[0]===6||c[0]===2)){r=0;continue}if(c[0]===3&&(!i||c[1]>i[0]&&c[1]=e.length&&(e=void 0),{value:e&&e[o++],done:!e}}};throw new TypeError(t?"Object is not iterable.":"Symbol.iterator is not defined.")}function z(e,t){var r=typeof Symbol=="function"&&e[Symbol.iterator];if(!r)return e;var o=r.call(e),n,i=[],a;try{for(;(t===void 0||t-- >0)&&!(n=o.next()).done;)i.push(n.value)}catch(s){a={error:s}}finally{try{n&&!n.done&&(r=o.return)&&r.call(o)}finally{if(a)throw a.error}}return i}function q(e,t,r){if(r||arguments.length===2)for(var o=0,n=t.length,i;o1||p(d,L)})},y&&(n[d]=y(n[d])))}function p(d,y){try{c(o[d](y))}catch(L){u(i[0][3],L)}}function c(d){d.value instanceof nt?Promise.resolve(d.value.v).then(l,f):u(i[0][2],d)}function l(d){p("next",d)}function f(d){p("throw",d)}function u(d,y){d(y),i.shift(),i.length&&p(i[0][0],i[0][1])}}function uo(e){if(!Symbol.asyncIterator)throw new TypeError("Symbol.asyncIterator is not defined.");var t=e[Symbol.asyncIterator],r;return t?t.call(e):(e=typeof he=="function"?he(e):e[Symbol.iterator](),r={},o("next"),o("throw"),o("return"),r[Symbol.asyncIterator]=function(){return this},r);function o(i){r[i]=e[i]&&function(a){return new Promise(function(s,p){a=e[i](a),n(s,p,a.done,a.value)})}}function n(i,a,s,p){Promise.resolve(p).then(function(c){i({value:c,done:s})},a)}}function H(e){return typeof e=="function"}function ut(e){var t=function(o){Error.call(o),o.stack=new Error().stack},r=e(t);return r.prototype=Object.create(Error.prototype),r.prototype.constructor=r,r}var zt=ut(function(e){return function(r){e(this),this.message=r?r.length+` errors occurred during unsubscription: +`+r.map(function(o,n){return n+1+") "+o.toString()}).join(` + `):"",this.name="UnsubscriptionError",this.errors=r}});function Qe(e,t){if(e){var r=e.indexOf(t);0<=r&&e.splice(r,1)}}var Ue=(function(){function e(t){this.initialTeardown=t,this.closed=!1,this._parentage=null,this._finalizers=null}return e.prototype.unsubscribe=function(){var t,r,o,n,i;if(!this.closed){this.closed=!0;var a=this._parentage;if(a)if(this._parentage=null,Array.isArray(a))try{for(var s=he(a),p=s.next();!p.done;p=s.next()){var c=p.value;c.remove(this)}}catch(L){t={error:L}}finally{try{p&&!p.done&&(r=s.return)&&r.call(s)}finally{if(t)throw t.error}}else a.remove(this);var l=this.initialTeardown;if(H(l))try{l()}catch(L){i=L instanceof zt?L.errors:[L]}var f=this._finalizers;if(f){this._finalizers=null;try{for(var u=he(f),d=u.next();!d.done;d=u.next()){var y=d.value;try{ho(y)}catch(L){i=i!=null?i:[],L instanceof zt?i=q(q([],z(i)),z(L.errors)):i.push(L)}}}catch(L){o={error:L}}finally{try{d&&!d.done&&(n=u.return)&&n.call(u)}finally{if(o)throw o.error}}}if(i)throw new zt(i)}},e.prototype.add=function(t){var r;if(t&&t!==this)if(this.closed)ho(t);else{if(t instanceof e){if(t.closed||t._hasParent(this))return;t._addParent(this)}(this._finalizers=(r=this._finalizers)!==null&&r!==void 0?r:[]).push(t)}},e.prototype._hasParent=function(t){var r=this._parentage;return r===t||Array.isArray(r)&&r.includes(t)},e.prototype._addParent=function(t){var r=this._parentage;this._parentage=Array.isArray(r)?(r.push(t),r):r?[r,t]:t},e.prototype._removeParent=function(t){var r=this._parentage;r===t?this._parentage=null:Array.isArray(r)&&Qe(r,t)},e.prototype.remove=function(t){var r=this._finalizers;r&&Qe(r,t),t instanceof e&&t._removeParent(this)},e.EMPTY=(function(){var t=new e;return t.closed=!0,t})(),e})();var Tr=Ue.EMPTY;function Nt(e){return e instanceof Ue||e&&"closed"in e&&H(e.remove)&&H(e.add)&&H(e.unsubscribe)}function ho(e){H(e)?e():e.unsubscribe()}var Pe={onUnhandledError:null,onStoppedNotification:null,Promise:void 0,useDeprecatedSynchronousErrorHandling:!1,useDeprecatedNextContext:!1};var dt={setTimeout:function(e,t){for(var r=[],o=2;o0},enumerable:!1,configurable:!0}),t.prototype._trySubscribe=function(r){return this._throwIfClosed(),e.prototype._trySubscribe.call(this,r)},t.prototype._subscribe=function(r){return this._throwIfClosed(),this._checkFinalizedStatuses(r),this._innerSubscribe(r)},t.prototype._innerSubscribe=function(r){var o=this,n=this,i=n.hasError,a=n.isStopped,s=n.observers;return i||a?Tr:(this.currentObservers=null,s.push(r),new Ue(function(){o.currentObservers=null,Qe(s,r)}))},t.prototype._checkFinalizedStatuses=function(r){var o=this,n=o.hasError,i=o.thrownError,a=o.isStopped;n?r.error(i):a&&r.complete()},t.prototype.asObservable=function(){var r=new j;return r.source=this,r},t.create=function(r,o){return new To(r,o)},t})(j);var To=(function(e){oe(t,e);function t(r,o){var n=e.call(this)||this;return n.destination=r,n.source=o,n}return t.prototype.next=function(r){var o,n;(n=(o=this.destination)===null||o===void 0?void 0:o.next)===null||n===void 0||n.call(o,r)},t.prototype.error=function(r){var o,n;(n=(o=this.destination)===null||o===void 0?void 0:o.error)===null||n===void 0||n.call(o,r)},t.prototype.complete=function(){var r,o;(o=(r=this.destination)===null||r===void 0?void 0:r.complete)===null||o===void 0||o.call(r)},t.prototype._subscribe=function(r){var o,n;return(n=(o=this.source)===null||o===void 0?void 0:o.subscribe(r))!==null&&n!==void 0?n:Tr},t})(g);var _r=(function(e){oe(t,e);function t(r){var o=e.call(this)||this;return o._value=r,o}return Object.defineProperty(t.prototype,"value",{get:function(){return this.getValue()},enumerable:!1,configurable:!0}),t.prototype._subscribe=function(r){var o=e.prototype._subscribe.call(this,r);return!o.closed&&r.next(this._value),o},t.prototype.getValue=function(){var r=this,o=r.hasError,n=r.thrownError,i=r._value;if(o)throw n;return this._throwIfClosed(),i},t.prototype.next=function(r){e.prototype.next.call(this,this._value=r)},t})(g);var _t={now:function(){return(_t.delegate||Date).now()},delegate:void 0};var At=(function(e){oe(t,e);function t(r,o,n){r===void 0&&(r=1/0),o===void 0&&(o=1/0),n===void 0&&(n=_t);var i=e.call(this)||this;return i._bufferSize=r,i._windowTime=o,i._timestampProvider=n,i._buffer=[],i._infiniteTimeWindow=!0,i._infiniteTimeWindow=o===1/0,i._bufferSize=Math.max(1,r),i._windowTime=Math.max(1,o),i}return t.prototype.next=function(r){var o=this,n=o.isStopped,i=o._buffer,a=o._infiniteTimeWindow,s=o._timestampProvider,p=o._windowTime;n||(i.push(r),!a&&i.push(s.now()+p)),this._trimBuffer(),e.prototype.next.call(this,r)},t.prototype._subscribe=function(r){this._throwIfClosed(),this._trimBuffer();for(var o=this._innerSubscribe(r),n=this,i=n._infiniteTimeWindow,a=n._buffer,s=a.slice(),p=0;p0?e.prototype.schedule.call(this,r,o):(this.delay=o,this.state=r,this.scheduler.flush(this),this)},t.prototype.execute=function(r,o){return o>0||this.closed?e.prototype.execute.call(this,r,o):this._execute(r,o)},t.prototype.requestAsyncId=function(r,o,n){return n===void 0&&(n=0),n!=null&&n>0||n==null&&this.delay>0?e.prototype.requestAsyncId.call(this,r,o,n):(r.flush(this),0)},t})(gt);var Lo=(function(e){oe(t,e);function t(){return e!==null&&e.apply(this,arguments)||this}return t})(yt);var kr=new Lo(Oo);var Mo=(function(e){oe(t,e);function t(r,o){var n=e.call(this,r,o)||this;return n.scheduler=r,n.work=o,n}return t.prototype.requestAsyncId=function(r,o,n){return n===void 0&&(n=0),n!==null&&n>0?e.prototype.requestAsyncId.call(this,r,o,n):(r.actions.push(this),r._scheduled||(r._scheduled=vt.requestAnimationFrame(function(){return r.flush(void 0)})))},t.prototype.recycleAsyncId=function(r,o,n){var i;if(n===void 0&&(n=0),n!=null?n>0:this.delay>0)return e.prototype.recycleAsyncId.call(this,r,o,n);var a=r.actions;o!=null&&o===r._scheduled&&((i=a[a.length-1])===null||i===void 0?void 0:i.id)!==o&&(vt.cancelAnimationFrame(o),r._scheduled=void 0)},t})(gt);var _o=(function(e){oe(t,e);function t(){return e!==null&&e.apply(this,arguments)||this}return t.prototype.flush=function(r){this._active=!0;var o;r?o=r.id:(o=this._scheduled,this._scheduled=void 0);var n=this.actions,i;r=r||n.shift();do if(i=r.execute(r.state,r.delay))break;while((r=n[0])&&r.id===o&&n.shift());if(this._active=!1,i){for(;(r=n[0])&&r.id===o&&n.shift();)r.unsubscribe();throw i}},t})(yt);var me=new _o(Mo);var S=new j(function(e){return e.complete()});function Kt(e){return e&&H(e.schedule)}function Hr(e){return e[e.length-1]}function Xe(e){return H(Hr(e))?e.pop():void 0}function ke(e){return Kt(Hr(e))?e.pop():void 0}function Yt(e,t){return typeof Hr(e)=="number"?e.pop():t}var xt=(function(e){return e&&typeof e.length=="number"&&typeof e!="function"});function Bt(e){return H(e==null?void 0:e.then)}function Gt(e){return H(e[bt])}function Jt(e){return Symbol.asyncIterator&&H(e==null?void 0:e[Symbol.asyncIterator])}function Xt(e){return new TypeError("You provided "+(e!==null&&typeof e=="object"?"an invalid object":"'"+e+"'")+" where a stream was expected. You can provide an Observable, Promise, ReadableStream, Array, AsyncIterable, or Iterable.")}function Zi(){return typeof Symbol!="function"||!Symbol.iterator?"@@iterator":Symbol.iterator}var Zt=Zi();function er(e){return H(e==null?void 0:e[Zt])}function tr(e){return fo(this,arguments,function(){var r,o,n,i;return Dt(this,function(a){switch(a.label){case 0:r=e.getReader(),a.label=1;case 1:a.trys.push([1,,9,10]),a.label=2;case 2:return[4,nt(r.read())];case 3:return o=a.sent(),n=o.value,i=o.done,i?[4,nt(void 0)]:[3,5];case 4:return[2,a.sent()];case 5:return[4,nt(n)];case 6:return[4,a.sent()];case 7:return a.sent(),[3,2];case 8:return[3,10];case 9:return r.releaseLock(),[7];case 10:return[2]}})})}function rr(e){return H(e==null?void 0:e.getReader)}function U(e){if(e instanceof j)return e;if(e!=null){if(Gt(e))return ea(e);if(xt(e))return ta(e);if(Bt(e))return ra(e);if(Jt(e))return Ao(e);if(er(e))return oa(e);if(rr(e))return na(e)}throw Xt(e)}function ea(e){return new j(function(t){var r=e[bt]();if(H(r.subscribe))return r.subscribe(t);throw new TypeError("Provided object does not correctly implement Symbol.observable")})}function ta(e){return new j(function(t){for(var r=0;r=2;return function(o){return o.pipe(e?b(function(n,i){return e(n,i,o)}):le,Te(1),r?Ve(t):Qo(function(){return new nr}))}}function jr(e){return e<=0?function(){return S}:E(function(t,r){var o=[];t.subscribe(T(r,function(n){o.push(n),e=2,!0))}function pe(e){e===void 0&&(e={});var t=e.connector,r=t===void 0?function(){return new g}:t,o=e.resetOnError,n=o===void 0?!0:o,i=e.resetOnComplete,a=i===void 0?!0:i,s=e.resetOnRefCountZero,p=s===void 0?!0:s;return function(c){var l,f,u,d=0,y=!1,L=!1,X=function(){f==null||f.unsubscribe(),f=void 0},ee=function(){X(),l=u=void 0,y=L=!1},J=function(){var k=l;ee(),k==null||k.unsubscribe()};return E(function(k,ft){d++,!L&&!y&&X();var qe=u=u!=null?u:r();ft.add(function(){d--,d===0&&!L&&!y&&(f=Ur(J,p))}),qe.subscribe(ft),!l&&d>0&&(l=new at({next:function(Fe){return qe.next(Fe)},error:function(Fe){L=!0,X(),f=Ur(ee,n,Fe),qe.error(Fe)},complete:function(){y=!0,X(),f=Ur(ee,a),qe.complete()}}),U(k).subscribe(l))})(c)}}function Ur(e,t){for(var r=[],o=2;oe.next(document)),e}function P(e,t=document){return Array.from(t.querySelectorAll(e))}function R(e,t=document){let r=fe(e,t);if(typeof r=="undefined")throw new ReferenceError(`Missing element: expected "${e}" to be present`);return r}function fe(e,t=document){return t.querySelector(e)||void 0}function Ie(){var e,t,r,o;return(o=(r=(t=(e=document.activeElement)==null?void 0:e.shadowRoot)==null?void 0:t.activeElement)!=null?r:document.activeElement)!=null?o:void 0}var wa=O(h(document.body,"focusin"),h(document.body,"focusout")).pipe(_e(1),Q(void 0),m(()=>Ie()||document.body),G(1));function et(e){return wa.pipe(m(t=>e.contains(t)),K())}function Ht(e,t){return C(()=>O(h(e,"mouseenter").pipe(m(()=>!0)),h(e,"mouseleave").pipe(m(()=>!1))).pipe(t?kt(r=>Le(+!r*t)):le,Q(e.matches(":hover"))))}function Jo(e,t){if(typeof t=="string"||typeof t=="number")e.innerHTML+=t.toString();else if(t instanceof Node)e.appendChild(t);else if(Array.isArray(t))for(let r of t)Jo(e,r)}function x(e,t,...r){let o=document.createElement(e);if(t)for(let n of Object.keys(t))typeof t[n]!="undefined"&&(typeof t[n]!="boolean"?o.setAttribute(n,t[n]):o.setAttribute(n,""));for(let n of r)Jo(o,n);return o}function sr(e){if(e>999){let t=+((e-950)%1e3>99);return`${((e+1e-6)/1e3).toFixed(t)}k`}else return e.toString()}function wt(e){let t=x("script",{src:e});return C(()=>(document.head.appendChild(t),O(h(t,"load"),h(t,"error").pipe(v(()=>$r(()=>new ReferenceError(`Invalid script: ${e}`))))).pipe(m(()=>{}),_(()=>document.head.removeChild(t)),Te(1))))}var Xo=new g,Ta=C(()=>typeof ResizeObserver=="undefined"?wt("https://unpkg.com/resize-observer-polyfill"):I(void 0)).pipe(m(()=>new ResizeObserver(e=>e.forEach(t=>Xo.next(t)))),v(e=>O(Ye,I(e)).pipe(_(()=>e.disconnect()))),G(1));function ce(e){return{width:e.offsetWidth,height:e.offsetHeight}}function ge(e){let t=e;for(;t.clientWidth===0&&t.parentElement;)t=t.parentElement;return Ta.pipe(w(r=>r.observe(t)),v(r=>Xo.pipe(b(o=>o.target===t),_(()=>r.unobserve(t)))),m(()=>ce(e)),Q(ce(e)))}function Tt(e){return{width:e.scrollWidth,height:e.scrollHeight}}function cr(e){let t=e.parentElement;for(;t&&(e.scrollWidth<=t.scrollWidth&&e.scrollHeight<=t.scrollHeight);)t=(e=t).parentElement;return t?e:void 0}function Zo(e){let t=[],r=e.parentElement;for(;r;)(e.clientWidth>r.clientWidth||e.clientHeight>r.clientHeight)&&t.push(r),r=(e=r).parentElement;return t.length===0&&t.push(document.documentElement),t}function De(e){return{x:e.offsetLeft,y:e.offsetTop}}function en(e){let t=e.getBoundingClientRect();return{x:t.x+window.scrollX,y:t.y+window.scrollY}}function tn(e){return O(h(window,"load"),h(window,"resize")).pipe(Me(0,me),m(()=>De(e)),Q(De(e)))}function pr(e){return{x:e.scrollLeft,y:e.scrollTop}}function ze(e){return O(h(e,"scroll"),h(window,"scroll"),h(window,"resize")).pipe(Me(0,me),m(()=>pr(e)),Q(pr(e)))}var rn=new g,Sa=C(()=>I(new IntersectionObserver(e=>{for(let t of e)rn.next(t)},{threshold:0}))).pipe(v(e=>O(Ye,I(e)).pipe(_(()=>e.disconnect()))),G(1));function tt(e){return Sa.pipe(w(t=>t.observe(e)),v(t=>rn.pipe(b(({target:r})=>r===e),_(()=>t.unobserve(e)),m(({isIntersecting:r})=>r))))}function on(e,t=16){return ze(e).pipe(m(({y:r})=>{let o=ce(e),n=Tt(e);return r>=n.height-o.height-t}),K())}var lr={drawer:R("[data-md-toggle=drawer]"),search:R("[data-md-toggle=search]")};function nn(e){return lr[e].checked}function Je(e,t){lr[e].checked!==t&&lr[e].click()}function Ne(e){let t=lr[e];return h(t,"change").pipe(m(()=>t.checked),Q(t.checked))}function Oa(e,t){switch(e.constructor){case HTMLInputElement:return e.type==="radio"?/^Arrow/.test(t):!0;case HTMLSelectElement:case HTMLTextAreaElement:return!0;default:return e.isContentEditable}}function La(){return O(h(window,"compositionstart").pipe(m(()=>!0)),h(window,"compositionend").pipe(m(()=>!1))).pipe(Q(!1))}function an(){let e=h(window,"keydown").pipe(b(t=>!(t.metaKey||t.ctrlKey)),m(t=>({mode:nn("search")?"search":"global",type:t.key,claim(){t.preventDefault(),t.stopPropagation()}})),b(({mode:t,type:r})=>{if(t==="global"){let o=Ie();if(typeof o!="undefined")return!Oa(o,r)}return!0}),pe());return La().pipe(v(t=>t?S:e))}function ye(){return new URL(location.href)}function lt(e,t=!1){if(B("navigation.instant")&&!t){let r=x("a",{href:e.href});document.body.appendChild(r),r.click(),r.remove()}else location.href=e.href}function sn(){return new g}function cn(){return location.hash.slice(1)}function pn(e){let t=x("a",{href:e});t.addEventListener("click",r=>r.stopPropagation()),t.click()}function Ma(e){return O(h(window,"hashchange"),e).pipe(m(cn),Q(cn()),b(t=>t.length>0),G(1))}function ln(e){return Ma(e).pipe(m(t=>fe(`[id="${t}"]`)),b(t=>typeof t!="undefined"))}function $t(e){let t=matchMedia(e);return ir(r=>t.addListener(()=>r(t.matches))).pipe(Q(t.matches))}function mn(){let e=matchMedia("print");return O(h(window,"beforeprint").pipe(m(()=>!0)),h(window,"afterprint").pipe(m(()=>!1))).pipe(Q(e.matches))}function zr(e,t){return e.pipe(v(r=>r?t():S))}function Nr(e,t){return new j(r=>{let o=new XMLHttpRequest;return o.open("GET",`${e}`),o.responseType="blob",o.addEventListener("load",()=>{o.status>=200&&o.status<300?(r.next(o.response),r.complete()):r.error(new Error(o.statusText))}),o.addEventListener("error",()=>{r.error(new Error("Network error"))}),o.addEventListener("abort",()=>{r.complete()}),typeof(t==null?void 0:t.progress$)!="undefined"&&(o.addEventListener("progress",n=>{var i;if(n.lengthComputable)t.progress$.next(n.loaded/n.total*100);else{let a=(i=o.getResponseHeader("Content-Length"))!=null?i:0;t.progress$.next(n.loaded/+a*100)}}),t.progress$.next(5)),o.send(),()=>o.abort()})}function je(e,t){return Nr(e,t).pipe(v(r=>r.text()),m(r=>JSON.parse(r)),G(1))}function fn(e,t){let r=new DOMParser;return Nr(e,t).pipe(v(o=>o.text()),m(o=>r.parseFromString(o,"text/html")),G(1))}function un(e,t){let r=new DOMParser;return Nr(e,t).pipe(v(o=>o.text()),m(o=>r.parseFromString(o,"text/xml")),G(1))}function dn(){return{x:Math.max(0,scrollX),y:Math.max(0,scrollY)}}function hn(){return O(h(window,"scroll",{passive:!0}),h(window,"resize",{passive:!0})).pipe(m(dn),Q(dn()))}function bn(){return{width:innerWidth,height:innerHeight}}function vn(){return h(window,"resize",{passive:!0}).pipe(m(bn),Q(bn()))}function gn(){return N([hn(),vn()]).pipe(m(([e,t])=>({offset:e,size:t})),G(1))}function mr(e,{viewport$:t,header$:r}){let o=t.pipe(te("size")),n=N([o,r]).pipe(m(()=>De(e)));return N([r,t,n]).pipe(m(([{height:i},{offset:a,size:s},{x:p,y:c}])=>({offset:{x:a.x-p,y:a.y-c+i},size:s})))}function _a(e){return h(e,"message",t=>t.data)}function Aa(e){let t=new g;return t.subscribe(r=>e.postMessage(r)),t}function yn(e,t=new Worker(e)){let r=_a(t),o=Aa(t),n=new g;n.subscribe(o);let i=o.pipe(Z(),ie(!0));return n.pipe(Z(),Re(r.pipe(W(i))),pe())}var Ca=R("#__config"),St=JSON.parse(Ca.textContent);St.base=`${new URL(St.base,ye())}`;function xe(){return St}function B(e){return St.features.includes(e)}function Ee(e,t){return typeof t!="undefined"?St.translations[e].replace("#",t.toString()):St.translations[e]}function Se(e,t=document){return R(`[data-md-component=${e}]`,t)}function ae(e,t=document){return P(`[data-md-component=${e}]`,t)}function ka(e){let t=R(".md-typeset > :first-child",e);return h(t,"click",{once:!0}).pipe(m(()=>R(".md-typeset",e)),m(r=>({hash:__md_hash(r.innerHTML)})))}function xn(e){if(!B("announce.dismiss")||!e.childElementCount)return S;if(!e.hidden){let t=R(".md-typeset",e);__md_hash(t.innerHTML)===__md_get("__announce")&&(e.hidden=!0)}return C(()=>{let t=new g;return t.subscribe(({hash:r})=>{e.hidden=!0,__md_set("__announce",r)}),ka(e).pipe(w(r=>t.next(r)),_(()=>t.complete()),m(r=>$({ref:e},r)))})}function Ha(e,{target$:t}){return t.pipe(m(r=>({hidden:r!==e})))}function En(e,t){let r=new g;return r.subscribe(({hidden:o})=>{e.hidden=o}),Ha(e,t).pipe(w(o=>r.next(o)),_(()=>r.complete()),m(o=>$({ref:e},o)))}function Pt(e,t){return t==="inline"?x("div",{class:"md-tooltip md-tooltip--inline",id:e,role:"tooltip"},x("div",{class:"md-tooltip__inner md-typeset"})):x("div",{class:"md-tooltip",id:e,role:"tooltip"},x("div",{class:"md-tooltip__inner md-typeset"}))}function wn(...e){return x("div",{class:"md-tooltip2",role:"tooltip"},x("div",{class:"md-tooltip2__inner md-typeset"},e))}function Tn(e,t){if(t=t?`${t}_annotation_${e}`:void 0,t){let r=t?`#${t}`:void 0;return x("aside",{class:"md-annotation",tabIndex:0},Pt(t),x("a",{href:r,class:"md-annotation__index",tabIndex:-1},x("span",{"data-md-annotation-id":e})))}else return x("aside",{class:"md-annotation",tabIndex:0},Pt(t),x("span",{class:"md-annotation__index",tabIndex:-1},x("span",{"data-md-annotation-id":e})))}function Sn(e){return x("button",{class:"md-clipboard md-icon",title:Ee("clipboard.copy"),"data-clipboard-target":`#${e} > code`})}var Ln=Lt(qr());function Qr(e,t){let r=t&2,o=t&1,n=Object.keys(e.terms).filter(p=>!e.terms[p]).reduce((p,c)=>[...p,x("del",null,(0,Ln.default)(c))," "],[]).slice(0,-1),i=xe(),a=new URL(e.location,i.base);B("search.highlight")&&a.searchParams.set("h",Object.entries(e.terms).filter(([,p])=>p).reduce((p,[c])=>`${p} ${c}`.trim(),""));let{tags:s}=xe();return x("a",{href:`${a}`,class:"md-search-result__link",tabIndex:-1},x("article",{class:"md-search-result__article md-typeset","data-md-score":e.score.toFixed(2)},r>0&&x("div",{class:"md-search-result__icon md-icon"}),r>0&&x("h1",null,e.title),r<=0&&x("h2",null,e.title),o>0&&e.text.length>0&&e.text,e.tags&&x("nav",{class:"md-tags"},e.tags.map(p=>{let c=s?p in s?`md-tag-icon md-tag--${s[p]}`:"md-tag-icon":"";return x("span",{class:`md-tag ${c}`},p)})),o>0&&n.length>0&&x("p",{class:"md-search-result__terms"},Ee("search.result.term.missing"),": ",...n)))}function Mn(e){let t=e[0].score,r=[...e],o=xe(),n=r.findIndex(l=>!`${new URL(l.location,o.base)}`.includes("#")),[i]=r.splice(n,1),a=r.findIndex(l=>l.scoreQr(l,1)),...p.length?[x("details",{class:"md-search-result__more"},x("summary",{tabIndex:-1},x("div",null,p.length>0&&p.length===1?Ee("search.result.more.one"):Ee("search.result.more.other",p.length))),...p.map(l=>Qr(l,1)))]:[]];return x("li",{class:"md-search-result__item"},c)}function _n(e){return x("ul",{class:"md-source__facts"},Object.entries(e).map(([t,r])=>x("li",{class:`md-source__fact md-source__fact--${t}`},typeof r=="number"?sr(r):r)))}function Kr(e){let t=`tabbed-control tabbed-control--${e}`;return x("div",{class:t,hidden:!0},x("button",{class:"tabbed-button",tabIndex:-1,"aria-hidden":"true"}))}function An(e){return x("div",{class:"md-typeset__scrollwrap"},x("div",{class:"md-typeset__table"},e))}function Ra(e){var o;let t=xe(),r=new URL(`../${e.version}/`,t.base);return x("li",{class:"md-version__item"},x("a",{href:`${r}`,class:"md-version__link"},e.title,((o=t.version)==null?void 0:o.alias)&&e.aliases.length>0&&x("span",{class:"md-version__alias"},e.aliases[0])))}function Cn(e,t){var o;let r=xe();return e=e.filter(n=>{var i;return!((i=n.properties)!=null&&i.hidden)}),x("div",{class:"md-version"},x("button",{class:"md-version__current","aria-label":Ee("select.version")},t.title,((o=r.version)==null?void 0:o.alias)&&t.aliases.length>0&&x("span",{class:"md-version__alias"},t.aliases[0])),x("ul",{class:"md-version__list"},e.map(Ra)))}var Ia=0;function ja(e){let t=N([et(e),Ht(e)]).pipe(m(([o,n])=>o||n),K()),r=C(()=>Zo(e)).pipe(ne(ze),pt(1),He(t),m(()=>en(e)));return t.pipe(Ae(o=>o),v(()=>N([t,r])),m(([o,n])=>({active:o,offset:n})),pe())}function Fa(e,t){let{content$:r,viewport$:o}=t,n=`__tooltip2_${Ia++}`;return C(()=>{let i=new g,a=new _r(!1);i.pipe(Z(),ie(!1)).subscribe(a);let s=a.pipe(kt(c=>Le(+!c*250,kr)),K(),v(c=>c?r:S),w(c=>c.id=n),pe());N([i.pipe(m(({active:c})=>c)),s.pipe(v(c=>Ht(c,250)),Q(!1))]).pipe(m(c=>c.some(l=>l))).subscribe(a);let p=a.pipe(b(c=>c),re(s,o),m(([c,l,{size:f}])=>{let u=e.getBoundingClientRect(),d=u.width/2;if(l.role==="tooltip")return{x:d,y:8+u.height};if(u.y>=f.height/2){let{height:y}=ce(l);return{x:d,y:-16-y}}else return{x:d,y:16+u.height}}));return N([s,i,p]).subscribe(([c,{offset:l},f])=>{c.style.setProperty("--md-tooltip-host-x",`${l.x}px`),c.style.setProperty("--md-tooltip-host-y",`${l.y}px`),c.style.setProperty("--md-tooltip-x",`${f.x}px`),c.style.setProperty("--md-tooltip-y",`${f.y}px`),c.classList.toggle("md-tooltip2--top",f.y<0),c.classList.toggle("md-tooltip2--bottom",f.y>=0)}),a.pipe(b(c=>c),re(s,(c,l)=>l),b(c=>c.role==="tooltip")).subscribe(c=>{let l=ce(R(":scope > *",c));c.style.setProperty("--md-tooltip-width",`${l.width}px`),c.style.setProperty("--md-tooltip-tail","0px")}),a.pipe(K(),ve(me),re(s)).subscribe(([c,l])=>{l.classList.toggle("md-tooltip2--active",c)}),N([a.pipe(b(c=>c)),s]).subscribe(([c,l])=>{l.role==="dialog"?(e.setAttribute("aria-controls",n),e.setAttribute("aria-haspopup","dialog")):e.setAttribute("aria-describedby",n)}),a.pipe(b(c=>!c)).subscribe(()=>{e.removeAttribute("aria-controls"),e.removeAttribute("aria-describedby"),e.removeAttribute("aria-haspopup")}),ja(e).pipe(w(c=>i.next(c)),_(()=>i.complete()),m(c=>$({ref:e},c)))})}function mt(e,{viewport$:t},r=document.body){return Fa(e,{content$:new j(o=>{let n=e.title,i=wn(n);return o.next(i),e.removeAttribute("title"),r.append(i),()=>{i.remove(),e.setAttribute("title",n)}}),viewport$:t})}function Ua(e,t){let r=C(()=>N([tn(e),ze(t)])).pipe(m(([{x:o,y:n},i])=>{let{width:a,height:s}=ce(e);return{x:o-i.x+a/2,y:n-i.y+s/2}}));return et(e).pipe(v(o=>r.pipe(m(n=>({active:o,offset:n})),Te(+!o||1/0))))}function kn(e,t,{target$:r}){let[o,n]=Array.from(e.children);return C(()=>{let i=new g,a=i.pipe(Z(),ie(!0));return i.subscribe({next({offset:s}){e.style.setProperty("--md-tooltip-x",`${s.x}px`),e.style.setProperty("--md-tooltip-y",`${s.y}px`)},complete(){e.style.removeProperty("--md-tooltip-x"),e.style.removeProperty("--md-tooltip-y")}}),tt(e).pipe(W(a)).subscribe(s=>{e.toggleAttribute("data-md-visible",s)}),O(i.pipe(b(({active:s})=>s)),i.pipe(_e(250),b(({active:s})=>!s))).subscribe({next({active:s}){s?e.prepend(o):o.remove()},complete(){e.prepend(o)}}),i.pipe(Me(16,me)).subscribe(({active:s})=>{o.classList.toggle("md-tooltip--active",s)}),i.pipe(pt(125,me),b(()=>!!e.offsetParent),m(()=>e.offsetParent.getBoundingClientRect()),m(({x:s})=>s)).subscribe({next(s){s?e.style.setProperty("--md-tooltip-0",`${-s}px`):e.style.removeProperty("--md-tooltip-0")},complete(){e.style.removeProperty("--md-tooltip-0")}}),h(n,"click").pipe(W(a),b(s=>!(s.metaKey||s.ctrlKey))).subscribe(s=>{s.stopPropagation(),s.preventDefault()}),h(n,"mousedown").pipe(W(a),re(i)).subscribe(([s,{active:p}])=>{var c;if(s.button!==0||s.metaKey||s.ctrlKey)s.preventDefault();else if(p){s.preventDefault();let l=e.parentElement.closest(".md-annotation");l instanceof HTMLElement?l.focus():(c=Ie())==null||c.blur()}}),r.pipe(W(a),b(s=>s===o),Ge(125)).subscribe(()=>e.focus()),Ua(e,t).pipe(w(s=>i.next(s)),_(()=>i.complete()),m(s=>$({ref:e},s)))})}function Wa(e){return e.tagName==="CODE"?P(".c, .c1, .cm",e):[e]}function Va(e){let t=[];for(let r of Wa(e)){let o=[],n=document.createNodeIterator(r,NodeFilter.SHOW_TEXT);for(let i=n.nextNode();i;i=n.nextNode())o.push(i);for(let i of o){let a;for(;a=/(\(\d+\))(!)?/.exec(i.textContent);){let[,s,p]=a;if(typeof p=="undefined"){let c=i.splitText(a.index);i=c.splitText(s.length),t.push(c)}else{i.textContent=s,t.push(i);break}}}}return t}function Hn(e,t){t.append(...Array.from(e.childNodes))}function fr(e,t,{target$:r,print$:o}){let n=t.closest("[id]"),i=n==null?void 0:n.id,a=new Map;for(let s of Va(t)){let[,p]=s.textContent.match(/\((\d+)\)/);fe(`:scope > li:nth-child(${p})`,e)&&(a.set(p,Tn(p,i)),s.replaceWith(a.get(p)))}return a.size===0?S:C(()=>{let s=new g,p=s.pipe(Z(),ie(!0)),c=[];for(let[l,f]of a)c.push([R(".md-typeset",f),R(`:scope > li:nth-child(${l})`,e)]);return o.pipe(W(p)).subscribe(l=>{e.hidden=!l,e.classList.toggle("md-annotation-list",l);for(let[f,u]of c)l?Hn(f,u):Hn(u,f)}),O(...[...a].map(([,l])=>kn(l,t,{target$:r}))).pipe(_(()=>s.complete()),pe())})}function $n(e){if(e.nextElementSibling){let t=e.nextElementSibling;if(t.tagName==="OL")return t;if(t.tagName==="P"&&!t.children.length)return $n(t)}}function Pn(e,t){return C(()=>{let r=$n(e);return typeof r!="undefined"?fr(r,e,t):S})}var Rn=Lt(Br());var Da=0;function In(e){if(e.nextElementSibling){let t=e.nextElementSibling;if(t.tagName==="OL")return t;if(t.tagName==="P"&&!t.children.length)return In(t)}}function za(e){return ge(e).pipe(m(({width:t})=>({scrollable:Tt(e).width>t})),te("scrollable"))}function jn(e,t){let{matches:r}=matchMedia("(hover)"),o=C(()=>{let n=new g,i=n.pipe(jr(1));n.subscribe(({scrollable:c})=>{c&&r?e.setAttribute("tabindex","0"):e.removeAttribute("tabindex")});let a=[];if(Rn.default.isSupported()&&(e.closest(".copy")||B("content.code.copy")&&!e.closest(".no-copy"))){let c=e.closest("pre");c.id=`__code_${Da++}`;let l=Sn(c.id);c.insertBefore(l,e),B("content.tooltips")&&a.push(mt(l,{viewport$}))}let s=e.closest(".highlight");if(s instanceof HTMLElement){let c=In(s);if(typeof c!="undefined"&&(s.classList.contains("annotate")||B("content.code.annotate"))){let l=fr(c,e,t);a.push(ge(s).pipe(W(i),m(({width:f,height:u})=>f&&u),K(),v(f=>f?l:S)))}}return P(":scope > span[id]",e).length&&e.classList.add("md-code__content"),za(e).pipe(w(c=>n.next(c)),_(()=>n.complete()),m(c=>$({ref:e},c)),Re(...a))});return B("content.lazy")?tt(e).pipe(b(n=>n),Te(1),v(()=>o)):o}function Na(e,{target$:t,print$:r}){let o=!0;return O(t.pipe(m(n=>n.closest("details:not([open])")),b(n=>e===n),m(()=>({action:"open",reveal:!0}))),r.pipe(b(n=>n||!o),w(()=>o=e.open),m(n=>({action:n?"open":"close"}))))}function Fn(e,t){return C(()=>{let r=new g;return r.subscribe(({action:o,reveal:n})=>{e.toggleAttribute("open",o==="open"),n&&e.scrollIntoView()}),Na(e,t).pipe(w(o=>r.next(o)),_(()=>r.complete()),m(o=>$({ref:e},o)))})}var Un=".node circle,.node ellipse,.node path,.node polygon,.node rect{fill:var(--md-mermaid-node-bg-color);stroke:var(--md-mermaid-node-fg-color)}marker{fill:var(--md-mermaid-edge-color)!important}.edgeLabel .label rect{fill:#0000}.flowchartTitleText{fill:var(--md-mermaid-label-fg-color)}.label{color:var(--md-mermaid-label-fg-color);font-family:var(--md-mermaid-font-family)}.label foreignObject{line-height:normal;overflow:visible}.label div .edgeLabel{color:var(--md-mermaid-label-fg-color)}.edgeLabel,.edgeLabel p,.label div .edgeLabel{background-color:var(--md-mermaid-label-bg-color)}.edgeLabel,.edgeLabel p{fill:var(--md-mermaid-label-bg-color);color:var(--md-mermaid-edge-color)}.edgePath .path,.flowchart-link{stroke:var(--md-mermaid-edge-color)}.edgePath .arrowheadPath{fill:var(--md-mermaid-edge-color);stroke:none}.cluster rect{fill:var(--md-default-fg-color--lightest);stroke:var(--md-default-fg-color--lighter)}.cluster span{color:var(--md-mermaid-label-fg-color);font-family:var(--md-mermaid-font-family)}g #flowchart-circleEnd,g #flowchart-circleStart,g #flowchart-crossEnd,g #flowchart-crossStart,g #flowchart-pointEnd,g #flowchart-pointStart{stroke:none}.classDiagramTitleText{fill:var(--md-mermaid-label-fg-color)}g.classGroup line,g.classGroup rect{fill:var(--md-mermaid-node-bg-color);stroke:var(--md-mermaid-node-fg-color)}g.classGroup text{fill:var(--md-mermaid-label-fg-color);font-family:var(--md-mermaid-font-family)}.classLabel .box{fill:var(--md-mermaid-label-bg-color);background-color:var(--md-mermaid-label-bg-color);opacity:1}.classLabel .label{fill:var(--md-mermaid-label-fg-color);font-family:var(--md-mermaid-font-family)}.node .divider{stroke:var(--md-mermaid-node-fg-color)}.relation{stroke:var(--md-mermaid-edge-color)}.cardinality{fill:var(--md-mermaid-label-fg-color);font-family:var(--md-mermaid-font-family)}.cardinality text{fill:inherit!important}defs marker.marker.composition.class path,defs marker.marker.dependency.class path,defs marker.marker.extension.class path{fill:var(--md-mermaid-edge-color)!important;stroke:var(--md-mermaid-edge-color)!important}defs marker.marker.aggregation.class path{fill:var(--md-mermaid-label-bg-color)!important;stroke:var(--md-mermaid-edge-color)!important}.statediagramTitleText{fill:var(--md-mermaid-label-fg-color)}g.stateGroup rect{fill:var(--md-mermaid-node-bg-color);stroke:var(--md-mermaid-node-fg-color)}g.stateGroup .state-title{fill:var(--md-mermaid-label-fg-color)!important;font-family:var(--md-mermaid-font-family)}g.stateGroup .composit{fill:var(--md-mermaid-label-bg-color)}.nodeLabel,.nodeLabel p{color:var(--md-mermaid-label-fg-color);font-family:var(--md-mermaid-font-family)}a .nodeLabel{text-decoration:underline}.node circle.state-end,.node circle.state-start,.start-state{fill:var(--md-mermaid-edge-color);stroke:none}.end-state-inner,.end-state-outer{fill:var(--md-mermaid-edge-color)}.end-state-inner,.node circle.state-end{stroke:var(--md-mermaid-label-bg-color)}.transition{stroke:var(--md-mermaid-edge-color)}[id^=state-fork] rect,[id^=state-join] rect{fill:var(--md-mermaid-edge-color)!important;stroke:none!important}.statediagram-cluster.statediagram-cluster .inner{fill:var(--md-default-bg-color)}.statediagram-cluster rect{fill:var(--md-mermaid-node-bg-color);stroke:var(--md-mermaid-node-fg-color)}.statediagram-state rect.divider{fill:var(--md-default-fg-color--lightest);stroke:var(--md-default-fg-color--lighter)}defs #statediagram-barbEnd{stroke:var(--md-mermaid-edge-color)}[id^=entity] path,[id^=entity] rect{fill:var(--md-default-bg-color)}.relationshipLine{stroke:var(--md-mermaid-edge-color)}defs .marker.oneOrMore.er *,defs .marker.onlyOne.er *,defs .marker.zeroOrMore.er *,defs .marker.zeroOrOne.er *{stroke:var(--md-mermaid-edge-color)!important}text:not([class]):last-child{fill:var(--md-mermaid-label-fg-color)}.actor{fill:var(--md-mermaid-sequence-actor-bg-color);stroke:var(--md-mermaid-sequence-actor-border-color)}text.actor>tspan{fill:var(--md-mermaid-sequence-actor-fg-color);font-family:var(--md-mermaid-font-family)}line{stroke:var(--md-mermaid-sequence-actor-line-color)}.actor-man circle,.actor-man line{fill:var(--md-mermaid-sequence-actorman-bg-color);stroke:var(--md-mermaid-sequence-actorman-line-color)}.messageLine0,.messageLine1{stroke:var(--md-mermaid-sequence-message-line-color)}.note{fill:var(--md-mermaid-sequence-note-bg-color);stroke:var(--md-mermaid-sequence-note-border-color)}.loopText,.loopText>tspan,.messageText,.noteText>tspan{stroke:none;font-family:var(--md-mermaid-font-family)!important}.messageText{fill:var(--md-mermaid-sequence-message-fg-color)}.loopText,.loopText>tspan{fill:var(--md-mermaid-sequence-loop-fg-color)}.noteText>tspan{fill:var(--md-mermaid-sequence-note-fg-color)}#arrowhead path{fill:var(--md-mermaid-sequence-message-line-color);stroke:none}.loopLine{fill:var(--md-mermaid-sequence-loop-bg-color);stroke:var(--md-mermaid-sequence-loop-border-color)}.labelBox{fill:var(--md-mermaid-sequence-label-bg-color);stroke:none}.labelText,.labelText>span{fill:var(--md-mermaid-sequence-label-fg-color);font-family:var(--md-mermaid-font-family)}.sequenceNumber{fill:var(--md-mermaid-sequence-number-fg-color)}rect.rect{fill:var(--md-mermaid-sequence-box-bg-color);stroke:none}rect.rect+text.text{fill:var(--md-mermaid-sequence-box-fg-color)}defs #sequencenumber{fill:var(--md-mermaid-sequence-number-bg-color)!important}";var Gr,Qa=0;function Ka(){return typeof mermaid=="undefined"||mermaid instanceof Element?wt("https://unpkg.com/mermaid@11/dist/mermaid.min.js"):I(void 0)}function Wn(e){return e.classList.remove("mermaid"),Gr||(Gr=Ka().pipe(w(()=>mermaid.initialize({startOnLoad:!1,themeCSS:Un,sequence:{actorFontSize:"16px",messageFontSize:"16px",noteFontSize:"16px"}})),m(()=>{}),G(1))),Gr.subscribe(()=>co(null,null,function*(){e.classList.add("mermaid");let t=`__mermaid_${Qa++}`,r=x("div",{class:"mermaid"}),o=e.textContent,{svg:n,fn:i}=yield mermaid.render(t,o),a=r.attachShadow({mode:"closed"});a.innerHTML=n,e.replaceWith(r),i==null||i(a)})),Gr.pipe(m(()=>({ref:e})))}var Vn=x("table");function Dn(e){return e.replaceWith(Vn),Vn.replaceWith(An(e)),I({ref:e})}function Ya(e){let t=e.find(r=>r.checked)||e[0];return O(...e.map(r=>h(r,"change").pipe(m(()=>R(`label[for="${r.id}"]`))))).pipe(Q(R(`label[for="${t.id}"]`)),m(r=>({active:r})))}function zn(e,{viewport$:t,target$:r}){let o=R(".tabbed-labels",e),n=P(":scope > input",e),i=Kr("prev");e.append(i);let a=Kr("next");return e.append(a),C(()=>{let s=new g,p=s.pipe(Z(),ie(!0));N([s,ge(e),tt(e)]).pipe(W(p),Me(1,me)).subscribe({next([{active:c},l]){let f=De(c),{width:u}=ce(c);e.style.setProperty("--md-indicator-x",`${f.x}px`),e.style.setProperty("--md-indicator-width",`${u}px`);let d=pr(o);(f.xd.x+l.width)&&o.scrollTo({left:Math.max(0,f.x-16),behavior:"smooth"})},complete(){e.style.removeProperty("--md-indicator-x"),e.style.removeProperty("--md-indicator-width")}}),N([ze(o),ge(o)]).pipe(W(p)).subscribe(([c,l])=>{let f=Tt(o);i.hidden=c.x<16,a.hidden=c.x>f.width-l.width-16}),O(h(i,"click").pipe(m(()=>-1)),h(a,"click").pipe(m(()=>1))).pipe(W(p)).subscribe(c=>{let{width:l}=ce(o);o.scrollBy({left:l*c,behavior:"smooth"})}),r.pipe(W(p),b(c=>n.includes(c))).subscribe(c=>c.click()),o.classList.add("tabbed-labels--linked");for(let c of n){let l=R(`label[for="${c.id}"]`);l.replaceChildren(x("a",{href:`#${l.htmlFor}`,tabIndex:-1},...Array.from(l.childNodes))),h(l.firstElementChild,"click").pipe(W(p),b(f=>!(f.metaKey||f.ctrlKey)),w(f=>{f.preventDefault(),f.stopPropagation()})).subscribe(()=>{history.replaceState({},"",`#${l.htmlFor}`),l.click()})}return B("content.tabs.link")&&s.pipe(Ce(1),re(t)).subscribe(([{active:c},{offset:l}])=>{let f=c.innerText.trim();if(c.hasAttribute("data-md-switching"))c.removeAttribute("data-md-switching");else{let u=e.offsetTop-l.y;for(let y of P("[data-tabs]"))for(let L of P(":scope > input",y)){let X=R(`label[for="${L.id}"]`);if(X!==c&&X.innerText.trim()===f){X.setAttribute("data-md-switching",""),L.click();break}}window.scrollTo({top:e.offsetTop-u});let d=__md_get("__tabs")||[];__md_set("__tabs",[...new Set([f,...d])])}}),s.pipe(W(p)).subscribe(()=>{for(let c of P("audio, video",e))c.offsetWidth&&c.autoplay?c.play().catch(()=>{}):c.pause()}),Ya(n).pipe(w(c=>s.next(c)),_(()=>s.complete()),m(c=>$({ref:e},c)))}).pipe(Ke(se))}function Nn(e,{viewport$:t,target$:r,print$:o}){return O(...P(".annotate:not(.highlight)",e).map(n=>Pn(n,{target$:r,print$:o})),...P("pre:not(.mermaid) > code",e).map(n=>jn(n,{target$:r,print$:o})),...P("pre.mermaid",e).map(n=>Wn(n)),...P("table:not([class])",e).map(n=>Dn(n)),...P("details",e).map(n=>Fn(n,{target$:r,print$:o})),...P("[data-tabs]",e).map(n=>zn(n,{viewport$:t,target$:r})),...P("[title]",e).filter(()=>B("content.tooltips")).map(n=>mt(n,{viewport$:t})))}function Ba(e,{alert$:t}){return t.pipe(v(r=>O(I(!0),I(!1).pipe(Ge(2e3))).pipe(m(o=>({message:r,active:o})))))}function qn(e,t){let r=R(".md-typeset",e);return C(()=>{let o=new g;return o.subscribe(({message:n,active:i})=>{e.classList.toggle("md-dialog--active",i),r.textContent=n}),Ba(e,t).pipe(w(n=>o.next(n)),_(()=>o.complete()),m(n=>$({ref:e},n)))})}var Ga=0;function Ja(e,t){document.body.append(e);let{width:r}=ce(e);e.style.setProperty("--md-tooltip-width",`${r}px`),e.remove();let o=cr(t),n=typeof o!="undefined"?ze(o):I({x:0,y:0}),i=O(et(t),Ht(t)).pipe(K());return N([i,n]).pipe(m(([a,s])=>{let{x:p,y:c}=De(t),l=ce(t),f=t.closest("table");return f&&t.parentElement&&(p+=f.offsetLeft+t.parentElement.offsetLeft,c+=f.offsetTop+t.parentElement.offsetTop),{active:a,offset:{x:p-s.x+l.width/2-r/2,y:c-s.y+l.height+8}}}))}function Qn(e){let t=e.title;if(!t.length)return S;let r=`__tooltip_${Ga++}`,o=Pt(r,"inline"),n=R(".md-typeset",o);return n.innerHTML=t,C(()=>{let i=new g;return i.subscribe({next({offset:a}){o.style.setProperty("--md-tooltip-x",`${a.x}px`),o.style.setProperty("--md-tooltip-y",`${a.y}px`)},complete(){o.style.removeProperty("--md-tooltip-x"),o.style.removeProperty("--md-tooltip-y")}}),O(i.pipe(b(({active:a})=>a)),i.pipe(_e(250),b(({active:a})=>!a))).subscribe({next({active:a}){a?(e.insertAdjacentElement("afterend",o),e.setAttribute("aria-describedby",r),e.removeAttribute("title")):(o.remove(),e.removeAttribute("aria-describedby"),e.setAttribute("title",t))},complete(){o.remove(),e.removeAttribute("aria-describedby"),e.setAttribute("title",t)}}),i.pipe(Me(16,me)).subscribe(({active:a})=>{o.classList.toggle("md-tooltip--active",a)}),i.pipe(pt(125,me),b(()=>!!e.offsetParent),m(()=>e.offsetParent.getBoundingClientRect()),m(({x:a})=>a)).subscribe({next(a){a?o.style.setProperty("--md-tooltip-0",`${-a}px`):o.style.removeProperty("--md-tooltip-0")},complete(){o.style.removeProperty("--md-tooltip-0")}}),Ja(o,e).pipe(w(a=>i.next(a)),_(()=>i.complete()),m(a=>$({ref:e},a)))}).pipe(Ke(se))}function Xa({viewport$:e}){if(!B("header.autohide"))return I(!1);let t=e.pipe(m(({offset:{y:n}})=>n),Be(2,1),m(([n,i])=>[nMath.abs(i-n.y)>100),m(([,[n]])=>n),K()),o=Ne("search");return N([e,o]).pipe(m(([{offset:n},i])=>n.y>400&&!i),K(),v(n=>n?r:I(!1)),Q(!1))}function Kn(e,t){return C(()=>N([ge(e),Xa(t)])).pipe(m(([{height:r},o])=>({height:r,hidden:o})),K((r,o)=>r.height===o.height&&r.hidden===o.hidden),G(1))}function Yn(e,{header$:t,main$:r}){return C(()=>{let o=new g,n=o.pipe(Z(),ie(!0));o.pipe(te("active"),He(t)).subscribe(([{active:a},{hidden:s}])=>{e.classList.toggle("md-header--shadow",a&&!s),e.hidden=s});let i=ue(P("[title]",e)).pipe(b(()=>B("content.tooltips")),ne(a=>Qn(a)));return r.subscribe(o),t.pipe(W(n),m(a=>$({ref:e},a)),Re(i.pipe(W(n))))})}function Za(e,{viewport$:t,header$:r}){return mr(e,{viewport$:t,header$:r}).pipe(m(({offset:{y:o}})=>{let{height:n}=ce(e);return{active:n>0&&o>=n}}),te("active"))}function Bn(e,t){return C(()=>{let r=new g;r.subscribe({next({active:n}){e.classList.toggle("md-header__title--active",n)},complete(){e.classList.remove("md-header__title--active")}});let o=fe(".md-content h1");return typeof o=="undefined"?S:Za(o,t).pipe(w(n=>r.next(n)),_(()=>r.complete()),m(n=>$({ref:e},n)))})}function Gn(e,{viewport$:t,header$:r}){let o=r.pipe(m(({height:i})=>i),K()),n=o.pipe(v(()=>ge(e).pipe(m(({height:i})=>({top:e.offsetTop,bottom:e.offsetTop+i})),te("bottom"))));return N([o,n,t]).pipe(m(([i,{top:a,bottom:s},{offset:{y:p},size:{height:c}}])=>(c=Math.max(0,c-Math.max(0,a-p,i)-Math.max(0,c+p-s)),{offset:a-i,height:c,active:a-i<=p})),K((i,a)=>i.offset===a.offset&&i.height===a.height&&i.active===a.active))}function es(e){let t=__md_get("__palette")||{index:e.findIndex(o=>matchMedia(o.getAttribute("data-md-color-media")).matches)},r=Math.max(0,Math.min(t.index,e.length-1));return I(...e).pipe(ne(o=>h(o,"change").pipe(m(()=>o))),Q(e[r]),m(o=>({index:e.indexOf(o),color:{media:o.getAttribute("data-md-color-media"),scheme:o.getAttribute("data-md-color-scheme"),primary:o.getAttribute("data-md-color-primary"),accent:o.getAttribute("data-md-color-accent")}})),G(1))}function Jn(e){let t=P("input",e),r=x("meta",{name:"theme-color"});document.head.appendChild(r);let o=x("meta",{name:"color-scheme"});document.head.appendChild(o);let n=$t("(prefers-color-scheme: light)");return C(()=>{let i=new g;return i.subscribe(a=>{if(document.body.setAttribute("data-md-color-switching",""),a.color.media==="(prefers-color-scheme)"){let s=matchMedia("(prefers-color-scheme: light)"),p=document.querySelector(s.matches?"[data-md-color-media='(prefers-color-scheme: light)']":"[data-md-color-media='(prefers-color-scheme: dark)']");a.color.scheme=p.getAttribute("data-md-color-scheme"),a.color.primary=p.getAttribute("data-md-color-primary"),a.color.accent=p.getAttribute("data-md-color-accent")}for(let[s,p]of Object.entries(a.color))document.body.setAttribute(`data-md-color-${s}`,p);for(let s=0;sa.key==="Enter"),re(i,(a,s)=>s)).subscribe(({index:a})=>{a=(a+1)%t.length,t[a].click(),t[a].focus()}),i.pipe(m(()=>{let a=Se("header"),s=window.getComputedStyle(a);return o.content=s.colorScheme,s.backgroundColor.match(/\d+/g).map(p=>(+p).toString(16).padStart(2,"0")).join("")})).subscribe(a=>r.content=`#${a}`),i.pipe(ve(se)).subscribe(()=>{document.body.removeAttribute("data-md-color-switching")}),es(t).pipe(W(n.pipe(Ce(1))),ct(),w(a=>i.next(a)),_(()=>i.complete()),m(a=>$({ref:e},a)))})}function Xn(e,{progress$:t}){return C(()=>{let r=new g;return r.subscribe(({value:o})=>{e.style.setProperty("--md-progress-value",`${o}`)}),t.pipe(w(o=>r.next({value:o})),_(()=>r.complete()),m(o=>({ref:e,value:o})))})}var Jr=Lt(Br());function ts(e){e.setAttribute("data-md-copying","");let t=e.closest("[data-copy]"),r=t?t.getAttribute("data-copy"):e.innerText;return e.removeAttribute("data-md-copying"),r.trimEnd()}function Zn({alert$:e}){Jr.default.isSupported()&&new j(t=>{new Jr.default("[data-clipboard-target], [data-clipboard-text]",{text:r=>r.getAttribute("data-clipboard-text")||ts(R(r.getAttribute("data-clipboard-target")))}).on("success",r=>t.next(r))}).pipe(w(t=>{t.trigger.focus()}),m(()=>Ee("clipboard.copied"))).subscribe(e)}function ei(e,t){return e.protocol=t.protocol,e.hostname=t.hostname,e}function rs(e,t){let r=new Map;for(let o of P("url",e)){let n=R("loc",o),i=[ei(new URL(n.textContent),t)];r.set(`${i[0]}`,i);for(let a of P("[rel=alternate]",o)){let s=a.getAttribute("href");s!=null&&i.push(ei(new URL(s),t))}}return r}function ur(e){return un(new URL("sitemap.xml",e)).pipe(m(t=>rs(t,new URL(e))),de(()=>I(new Map)))}function os(e,t){if(!(e.target instanceof Element))return S;let r=e.target.closest("a");if(r===null)return S;if(r.target||e.metaKey||e.ctrlKey)return S;let o=new URL(r.href);return o.search=o.hash="",t.has(`${o}`)?(e.preventDefault(),I(new URL(r.href))):S}function ti(e){let t=new Map;for(let r of P(":scope > *",e.head))t.set(r.outerHTML,r);return t}function ri(e){for(let t of P("[href], [src]",e))for(let r of["href","src"]){let o=t.getAttribute(r);if(o&&!/^(?:[a-z]+:)?\/\//i.test(o)){t[r]=t[r];break}}return I(e)}function ns(e){for(let o of["[data-md-component=announce]","[data-md-component=container]","[data-md-component=header-topic]","[data-md-component=outdated]","[data-md-component=logo]","[data-md-component=skip]",...B("navigation.tabs.sticky")?["[data-md-component=tabs]"]:[]]){let n=fe(o),i=fe(o,e);typeof n!="undefined"&&typeof i!="undefined"&&n.replaceWith(i)}let t=ti(document);for(let[o,n]of ti(e))t.has(o)?t.delete(o):document.head.appendChild(n);for(let o of t.values()){let n=o.getAttribute("name");n!=="theme-color"&&n!=="color-scheme"&&o.remove()}let r=Se("container");return We(P("script",r)).pipe(v(o=>{let n=e.createElement("script");if(o.src){for(let i of o.getAttributeNames())n.setAttribute(i,o.getAttribute(i));return o.replaceWith(n),new j(i=>{n.onload=()=>i.complete()})}else return n.textContent=o.textContent,o.replaceWith(n),S}),Z(),ie(document))}function oi({location$:e,viewport$:t,progress$:r}){let o=xe();if(location.protocol==="file:")return S;let n=ur(o.base);I(document).subscribe(ri);let i=h(document.body,"click").pipe(He(n),v(([p,c])=>os(p,c)),pe()),a=h(window,"popstate").pipe(m(ye),pe());i.pipe(re(t)).subscribe(([p,{offset:c}])=>{history.replaceState(c,""),history.pushState(null,"",p)}),O(i,a).subscribe(e);let s=e.pipe(te("pathname"),v(p=>fn(p,{progress$:r}).pipe(de(()=>(lt(p,!0),S)))),v(ri),v(ns),pe());return O(s.pipe(re(e,(p,c)=>c)),s.pipe(v(()=>e),te("hash")),e.pipe(K((p,c)=>p.pathname===c.pathname&&p.hash===c.hash),v(()=>i),w(()=>history.back()))).subscribe(p=>{var c,l;history.state!==null||!p.hash?window.scrollTo(0,(l=(c=history.state)==null?void 0:c.y)!=null?l:0):(history.scrollRestoration="auto",pn(p.hash),history.scrollRestoration="manual")}),e.subscribe(()=>{history.scrollRestoration="manual"}),h(window,"beforeunload").subscribe(()=>{history.scrollRestoration="auto"}),t.pipe(te("offset"),_e(100)).subscribe(({offset:p})=>{history.replaceState(p,"")}),s}var ni=Lt(qr());function ii(e){let t=e.separator.split("|").map(n=>n.replace(/(\(\?[!=<][^)]+\))/g,"").length===0?"\uFFFD":n).join("|"),r=new RegExp(t,"img"),o=(n,i,a)=>`${i}${a}`;return n=>{n=n.replace(/[\s*+\-:~^]+/g," ").replace(/&/g,"&").trim();let i=new RegExp(`(^|${e.separator}|)(${n.replace(/[|\\{}()[\]^$+*?.-]/g,"\\$&").replace(r,"|")})`,"img");return a=>(0,ni.default)(a).replace(i,o).replace(/<\/mark>(\s+)]*>/img,"$1")}}function It(e){return e.type===1}function dr(e){return e.type===3}function ai(e,t){let r=yn(e);return O(I(location.protocol!=="file:"),Ne("search")).pipe(Ae(o=>o),v(()=>t)).subscribe(({config:o,docs:n})=>r.next({type:0,data:{config:o,docs:n,options:{suggest:B("search.suggest")}}})),r}function si(e){var l;let{selectedVersionSitemap:t,selectedVersionBaseURL:r,currentLocation:o,currentBaseURL:n}=e,i=(l=Xr(n))==null?void 0:l.pathname;if(i===void 0)return;let a=ss(o.pathname,i);if(a===void 0)return;let s=ps(t.keys());if(!t.has(s))return;let p=Xr(a,s);if(!p||!t.has(p.href))return;let c=Xr(a,r);if(c)return c.hash=o.hash,c.search=o.search,c}function Xr(e,t){try{return new URL(e,t)}catch(r){return}}function ss(e,t){if(e.startsWith(t))return e.slice(t.length)}function cs(e,t){let r=Math.min(e.length,t.length),o;for(o=0;oS)),o=r.pipe(m(n=>{let[,i]=t.base.match(/([^/]+)\/?$/);return n.find(({version:a,aliases:s})=>a===i||s.includes(i))||n[0]}));r.pipe(m(n=>new Map(n.map(i=>[`${new URL(`../${i.version}/`,t.base)}`,i]))),v(n=>h(document.body,"click").pipe(b(i=>!i.metaKey&&!i.ctrlKey),re(o),v(([i,a])=>{if(i.target instanceof Element){let s=i.target.closest("a");if(s&&!s.target&&n.has(s.href)){let p=s.href;return!i.target.closest(".md-version")&&n.get(p)===a?S:(i.preventDefault(),I(new URL(p)))}}return S}),v(i=>ur(i).pipe(m(a=>{var s;return(s=si({selectedVersionSitemap:a,selectedVersionBaseURL:i,currentLocation:ye(),currentBaseURL:t.base}))!=null?s:i})))))).subscribe(n=>lt(n,!0)),N([r,o]).subscribe(([n,i])=>{R(".md-header__topic").appendChild(Cn(n,i))}),e.pipe(v(()=>o)).subscribe(n=>{var s;let i=new URL(t.base),a=__md_get("__outdated",sessionStorage,i);if(a===null){a=!0;let p=((s=t.version)==null?void 0:s.default)||"latest";Array.isArray(p)||(p=[p]);e:for(let c of p)for(let l of n.aliases.concat(n.version))if(new RegExp(c,"i").test(l)){a=!1;break e}__md_set("__outdated",a,sessionStorage,i)}if(a)for(let p of ae("outdated"))p.hidden=!1})}function ls(e,{worker$:t}){let{searchParams:r}=ye();r.has("q")&&(Je("search",!0),e.value=r.get("q"),e.focus(),Ne("search").pipe(Ae(i=>!i)).subscribe(()=>{let i=ye();i.searchParams.delete("q"),history.replaceState({},"",`${i}`)}));let o=et(e),n=O(t.pipe(Ae(It)),h(e,"keyup"),o).pipe(m(()=>e.value),K());return N([n,o]).pipe(m(([i,a])=>({value:i,focus:a})),G(1))}function pi(e,{worker$:t}){let r=new g,o=r.pipe(Z(),ie(!0));N([t.pipe(Ae(It)),r],(i,a)=>a).pipe(te("value")).subscribe(({value:i})=>t.next({type:2,data:i})),r.pipe(te("focus")).subscribe(({focus:i})=>{i&&Je("search",i)}),h(e.form,"reset").pipe(W(o)).subscribe(()=>e.focus());let n=R("header [for=__search]");return h(n,"click").subscribe(()=>e.focus()),ls(e,{worker$:t}).pipe(w(i=>r.next(i)),_(()=>r.complete()),m(i=>$({ref:e},i)),G(1))}function li(e,{worker$:t,query$:r}){let o=new g,n=on(e.parentElement).pipe(b(Boolean)),i=e.parentElement,a=R(":scope > :first-child",e),s=R(":scope > :last-child",e);Ne("search").subscribe(l=>{s.setAttribute("role",l?"list":"presentation"),s.hidden=!l}),o.pipe(re(r),Wr(t.pipe(Ae(It)))).subscribe(([{items:l},{value:f}])=>{switch(l.length){case 0:a.textContent=f.length?Ee("search.result.none"):Ee("search.result.placeholder");break;case 1:a.textContent=Ee("search.result.one");break;default:let u=sr(l.length);a.textContent=Ee("search.result.other",u)}});let p=o.pipe(w(()=>s.innerHTML=""),v(({items:l})=>O(I(...l.slice(0,10)),I(...l.slice(10)).pipe(Be(4),Dr(n),v(([f])=>f)))),m(Mn),pe());return p.subscribe(l=>s.appendChild(l)),p.pipe(ne(l=>{let f=fe("details",l);return typeof f=="undefined"?S:h(f,"toggle").pipe(W(o),m(()=>f))})).subscribe(l=>{l.open===!1&&l.offsetTop<=i.scrollTop&&i.scrollTo({top:l.offsetTop})}),t.pipe(b(dr),m(({data:l})=>l)).pipe(w(l=>o.next(l)),_(()=>o.complete()),m(l=>$({ref:e},l)))}function ms(e,{query$:t}){return t.pipe(m(({value:r})=>{let o=ye();return o.hash="",r=r.replace(/\s+/g,"+").replace(/&/g,"%26").replace(/=/g,"%3D"),o.search=`q=${r}`,{url:o}}))}function mi(e,t){let r=new g,o=r.pipe(Z(),ie(!0));return r.subscribe(({url:n})=>{e.setAttribute("data-clipboard-text",e.href),e.href=`${n}`}),h(e,"click").pipe(W(o)).subscribe(n=>n.preventDefault()),ms(e,t).pipe(w(n=>r.next(n)),_(()=>r.complete()),m(n=>$({ref:e},n)))}function fi(e,{worker$:t,keyboard$:r}){let o=new g,n=Se("search-query"),i=O(h(n,"keydown"),h(n,"focus")).pipe(ve(se),m(()=>n.value),K());return o.pipe(He(i),m(([{suggest:s},p])=>{let c=p.split(/([\s-]+)/);if(s!=null&&s.length&&c[c.length-1]){let l=s[s.length-1];l.startsWith(c[c.length-1])&&(c[c.length-1]=l)}else c.length=0;return c})).subscribe(s=>e.innerHTML=s.join("").replace(/\s/g," ")),r.pipe(b(({mode:s})=>s==="search")).subscribe(s=>{switch(s.type){case"ArrowRight":e.innerText.length&&n.selectionStart===n.value.length&&(n.value=e.innerText);break}}),t.pipe(b(dr),m(({data:s})=>s)).pipe(w(s=>o.next(s)),_(()=>o.complete()),m(()=>({ref:e})))}function ui(e,{index$:t,keyboard$:r}){let o=xe();try{let n=ai(o.search,t),i=Se("search-query",e),a=Se("search-result",e);h(e,"click").pipe(b(({target:p})=>p instanceof Element&&!!p.closest("a"))).subscribe(()=>Je("search",!1)),r.pipe(b(({mode:p})=>p==="search")).subscribe(p=>{let c=Ie();switch(p.type){case"Enter":if(c===i){let l=new Map;for(let f of P(":first-child [href]",a)){let u=f.firstElementChild;l.set(f,parseFloat(u.getAttribute("data-md-score")))}if(l.size){let[[f]]=[...l].sort(([,u],[,d])=>d-u);f.click()}p.claim()}break;case"Escape":case"Tab":Je("search",!1),i.blur();break;case"ArrowUp":case"ArrowDown":if(typeof c=="undefined")i.focus();else{let l=[i,...P(":not(details) > [href], summary, details[open] [href]",a)],f=Math.max(0,(Math.max(0,l.indexOf(c))+l.length+(p.type==="ArrowUp"?-1:1))%l.length);l[f].focus()}p.claim();break;default:i!==Ie()&&i.focus()}}),r.pipe(b(({mode:p})=>p==="global")).subscribe(p=>{switch(p.type){case"f":case"s":case"/":i.focus(),i.select(),p.claim();break}});let s=pi(i,{worker$:n});return O(s,li(a,{worker$:n,query$:s})).pipe(Re(...ae("search-share",e).map(p=>mi(p,{query$:s})),...ae("search-suggest",e).map(p=>fi(p,{worker$:n,keyboard$:r}))))}catch(n){return e.hidden=!0,Ye}}function di(e,{index$:t,location$:r}){return N([t,r.pipe(Q(ye()),b(o=>!!o.searchParams.get("h")))]).pipe(m(([o,n])=>ii(o.config)(n.searchParams.get("h"))),m(o=>{var a;let n=new Map,i=document.createNodeIterator(e,NodeFilter.SHOW_TEXT);for(let s=i.nextNode();s;s=i.nextNode())if((a=s.parentElement)!=null&&a.offsetHeight){let p=s.textContent,c=o(p);c.length>p.length&&n.set(s,c)}for(let[s,p]of n){let{childNodes:c}=x("span",null,p);s.replaceWith(...Array.from(c))}return{ref:e,nodes:n}}))}function fs(e,{viewport$:t,main$:r}){let o=e.closest(".md-grid"),n=o.offsetTop-o.parentElement.offsetTop;return N([r,t]).pipe(m(([{offset:i,height:a},{offset:{y:s}}])=>(a=a+Math.min(n,Math.max(0,s-i))-n,{height:a,locked:s>=i+n})),K((i,a)=>i.height===a.height&&i.locked===a.locked))}function Zr(e,o){var n=o,{header$:t}=n,r=so(n,["header$"]);let i=R(".md-sidebar__scrollwrap",e),{y:a}=De(i);return C(()=>{let s=new g,p=s.pipe(Z(),ie(!0)),c=s.pipe(Me(0,me));return c.pipe(re(t)).subscribe({next([{height:l},{height:f}]){i.style.height=`${l-2*a}px`,e.style.top=`${f}px`},complete(){i.style.height="",e.style.top=""}}),c.pipe(Ae()).subscribe(()=>{for(let l of P(".md-nav__link--active[href]",e)){if(!l.clientHeight)continue;let f=l.closest(".md-sidebar__scrollwrap");if(typeof f!="undefined"){let u=l.offsetTop-f.offsetTop,{height:d}=ce(f);f.scrollTo({top:u-d/2})}}}),ue(P("label[tabindex]",e)).pipe(ne(l=>h(l,"click").pipe(ve(se),m(()=>l),W(p)))).subscribe(l=>{let f=R(`[id="${l.htmlFor}"]`);R(`[aria-labelledby="${l.id}"]`).setAttribute("aria-expanded",`${f.checked}`)}),fs(e,r).pipe(w(l=>s.next(l)),_(()=>s.complete()),m(l=>$({ref:e},l)))})}function hi(e,t){if(typeof t!="undefined"){let r=`https://api.github.com/repos/${e}/${t}`;return st(je(`${r}/releases/latest`).pipe(de(()=>S),m(o=>({version:o.tag_name})),Ve({})),je(r).pipe(de(()=>S),m(o=>({stars:o.stargazers_count,forks:o.forks_count})),Ve({}))).pipe(m(([o,n])=>$($({},o),n)))}else{let r=`https://api.github.com/users/${e}`;return je(r).pipe(m(o=>({repositories:o.public_repos})),Ve({}))}}function bi(e,t){let r=`https://${e}/api/v4/projects/${encodeURIComponent(t)}`;return st(je(`${r}/releases/permalink/latest`).pipe(de(()=>S),m(({tag_name:o})=>({version:o})),Ve({})),je(r).pipe(de(()=>S),m(({star_count:o,forks_count:n})=>({stars:o,forks:n})),Ve({}))).pipe(m(([o,n])=>$($({},o),n)))}function vi(e){let t=e.match(/^.+github\.com\/([^/]+)\/?([^/]+)?/i);if(t){let[,r,o]=t;return hi(r,o)}if(t=e.match(/^.+?([^/]*gitlab[^/]+)\/(.+?)\/?$/i),t){let[,r,o]=t;return bi(r,o)}return S}var us;function ds(e){return us||(us=C(()=>{let t=__md_get("__source",sessionStorage);if(t)return I(t);if(ae("consent").length){let o=__md_get("__consent");if(!(o&&o.github))return S}return vi(e.href).pipe(w(o=>__md_set("__source",o,sessionStorage)))}).pipe(de(()=>S),b(t=>Object.keys(t).length>0),m(t=>({facts:t})),G(1)))}function gi(e){let t=R(":scope > :last-child",e);return C(()=>{let r=new g;return r.subscribe(({facts:o})=>{t.appendChild(_n(o)),t.classList.add("md-source__repository--active")}),ds(e).pipe(w(o=>r.next(o)),_(()=>r.complete()),m(o=>$({ref:e},o)))})}function hs(e,{viewport$:t,header$:r}){return ge(document.body).pipe(v(()=>mr(e,{header$:r,viewport$:t})),m(({offset:{y:o}})=>({hidden:o>=10})),te("hidden"))}function yi(e,t){return C(()=>{let r=new g;return r.subscribe({next({hidden:o}){e.hidden=o},complete(){e.hidden=!1}}),(B("navigation.tabs.sticky")?I({hidden:!1}):hs(e,t)).pipe(w(o=>r.next(o)),_(()=>r.complete()),m(o=>$({ref:e},o)))})}function bs(e,{viewport$:t,header$:r}){let o=new Map,n=P(".md-nav__link",e);for(let s of n){let p=decodeURIComponent(s.hash.substring(1)),c=fe(`[id="${p}"]`);typeof c!="undefined"&&o.set(s,c)}let i=r.pipe(te("height"),m(({height:s})=>{let p=Se("main"),c=R(":scope > :first-child",p);return s+.8*(c.offsetTop-p.offsetTop)}),pe());return ge(document.body).pipe(te("height"),v(s=>C(()=>{let p=[];return I([...o].reduce((c,[l,f])=>{for(;p.length&&o.get(p[p.length-1]).tagName>=f.tagName;)p.pop();let u=f.offsetTop;for(;!u&&f.parentElement;)f=f.parentElement,u=f.offsetTop;let d=f.offsetParent;for(;d;d=d.offsetParent)u+=d.offsetTop;return c.set([...p=[...p,l]].reverse(),u)},new Map))}).pipe(m(p=>new Map([...p].sort(([,c],[,l])=>c-l))),He(i),v(([p,c])=>t.pipe(Fr(([l,f],{offset:{y:u},size:d})=>{let y=u+d.height>=Math.floor(s.height);for(;f.length;){let[,L]=f[0];if(L-c=u&&!y)f=[l.pop(),...f];else break}return[l,f]},[[],[...p]]),K((l,f)=>l[0]===f[0]&&l[1]===f[1])))))).pipe(m(([s,p])=>({prev:s.map(([c])=>c),next:p.map(([c])=>c)})),Q({prev:[],next:[]}),Be(2,1),m(([s,p])=>s.prev.length{let i=new g,a=i.pipe(Z(),ie(!0));if(i.subscribe(({prev:s,next:p})=>{for(let[c]of p)c.classList.remove("md-nav__link--passed"),c.classList.remove("md-nav__link--active");for(let[c,[l]]of s.entries())l.classList.add("md-nav__link--passed"),l.classList.toggle("md-nav__link--active",c===s.length-1)}),B("toc.follow")){let s=O(t.pipe(_e(1),m(()=>{})),t.pipe(_e(250),m(()=>"smooth")));i.pipe(b(({prev:p})=>p.length>0),He(o.pipe(ve(se))),re(s)).subscribe(([[{prev:p}],c])=>{let[l]=p[p.length-1];if(l.offsetHeight){let f=cr(l);if(typeof f!="undefined"){let u=l.offsetTop-f.offsetTop,{height:d}=ce(f);f.scrollTo({top:u-d/2,behavior:c})}}})}return B("navigation.tracking")&&t.pipe(W(a),te("offset"),_e(250),Ce(1),W(n.pipe(Ce(1))),ct({delay:250}),re(i)).subscribe(([,{prev:s}])=>{let p=ye(),c=s[s.length-1];if(c&&c.length){let[l]=c,{hash:f}=new URL(l.href);p.hash!==f&&(p.hash=f,history.replaceState({},"",`${p}`))}else p.hash="",history.replaceState({},"",`${p}`)}),bs(e,{viewport$:t,header$:r}).pipe(w(s=>i.next(s)),_(()=>i.complete()),m(s=>$({ref:e},s)))})}function vs(e,{viewport$:t,main$:r,target$:o}){let n=t.pipe(m(({offset:{y:a}})=>a),Be(2,1),m(([a,s])=>a>s&&s>0),K()),i=r.pipe(m(({active:a})=>a));return N([i,n]).pipe(m(([a,s])=>!(a&&s)),K(),W(o.pipe(Ce(1))),ie(!0),ct({delay:250}),m(a=>({hidden:a})))}function Ei(e,{viewport$:t,header$:r,main$:o,target$:n}){let i=new g,a=i.pipe(Z(),ie(!0));return i.subscribe({next({hidden:s}){e.hidden=s,s?(e.setAttribute("tabindex","-1"),e.blur()):e.removeAttribute("tabindex")},complete(){e.style.top="",e.hidden=!0,e.removeAttribute("tabindex")}}),r.pipe(W(a),te("height")).subscribe(({height:s})=>{e.style.top=`${s+16}px`}),h(e,"click").subscribe(s=>{s.preventDefault(),window.scrollTo({top:0})}),vs(e,{viewport$:t,main$:o,target$:n}).pipe(w(s=>i.next(s)),_(()=>i.complete()),m(s=>$({ref:e},s)))}function wi({document$:e,viewport$:t}){e.pipe(v(()=>P(".md-ellipsis")),ne(r=>tt(r).pipe(W(e.pipe(Ce(1))),b(o=>o),m(()=>r),Te(1))),b(r=>r.offsetWidth{let o=r.innerText,n=r.closest("a")||r;return n.title=o,B("content.tooltips")?mt(n,{viewport$:t}).pipe(W(e.pipe(Ce(1))),_(()=>n.removeAttribute("title"))):S})).subscribe(),B("content.tooltips")&&e.pipe(v(()=>P(".md-status")),ne(r=>mt(r,{viewport$:t}))).subscribe()}function Ti({document$:e,tablet$:t}){e.pipe(v(()=>P(".md-toggle--indeterminate")),w(r=>{r.indeterminate=!0,r.checked=!1}),ne(r=>h(r,"change").pipe(Vr(()=>r.classList.contains("md-toggle--indeterminate")),m(()=>r))),re(t)).subscribe(([r,o])=>{r.classList.remove("md-toggle--indeterminate"),o&&(r.checked=!1)})}function gs(){return/(iPad|iPhone|iPod)/.test(navigator.userAgent)}function Si({document$:e}){e.pipe(v(()=>P("[data-md-scrollfix]")),w(t=>t.removeAttribute("data-md-scrollfix")),b(gs),ne(t=>h(t,"touchstart").pipe(m(()=>t)))).subscribe(t=>{let r=t.scrollTop;r===0?t.scrollTop=1:r+t.offsetHeight===t.scrollHeight&&(t.scrollTop=r-1)})}function Oi({viewport$:e,tablet$:t}){N([Ne("search"),t]).pipe(m(([r,o])=>r&&!o),v(r=>I(r).pipe(Ge(r?400:100))),re(e)).subscribe(([r,{offset:{y:o}}])=>{if(r)document.body.setAttribute("data-md-scrolllock",""),document.body.style.top=`-${o}px`;else{let n=-1*parseInt(document.body.style.top,10);document.body.removeAttribute("data-md-scrolllock"),document.body.style.top="",n&&window.scrollTo(0,n)}})}Object.entries||(Object.entries=function(e){let t=[];for(let r of Object.keys(e))t.push([r,e[r]]);return t});Object.values||(Object.values=function(e){let t=[];for(let r of Object.keys(e))t.push(e[r]);return t});typeof Element!="undefined"&&(Element.prototype.scrollTo||(Element.prototype.scrollTo=function(e,t){typeof e=="object"?(this.scrollLeft=e.left,this.scrollTop=e.top):(this.scrollLeft=e,this.scrollTop=t)}),Element.prototype.replaceWith||(Element.prototype.replaceWith=function(...e){let t=this.parentNode;if(t){e.length===0&&t.removeChild(this);for(let r=e.length-1;r>=0;r--){let o=e[r];typeof o=="string"?o=document.createTextNode(o):o.parentNode&&o.parentNode.removeChild(o),r?t.insertBefore(this.previousSibling,o):t.replaceChild(o,this)}}}));function ys(){return location.protocol==="file:"?wt(`${new URL("search/search_index.js",eo.base)}`).pipe(m(()=>__index),G(1)):je(new URL("search/search_index.json",eo.base))}document.documentElement.classList.remove("no-js");document.documentElement.classList.add("js");var ot=Go(),Ft=sn(),Ot=ln(Ft),to=an(),Oe=gn(),hr=$t("(min-width: 60em)"),Mi=$t("(min-width: 76.25em)"),_i=mn(),eo=xe(),Ai=document.forms.namedItem("search")?ys():Ye,ro=new g;Zn({alert$:ro});var oo=new g;B("navigation.instant")&&oi({location$:Ft,viewport$:Oe,progress$:oo}).subscribe(ot);var Li;((Li=eo.version)==null?void 0:Li.provider)==="mike"&&ci({document$:ot});O(Ft,Ot).pipe(Ge(125)).subscribe(()=>{Je("drawer",!1),Je("search",!1)});to.pipe(b(({mode:e})=>e==="global")).subscribe(e=>{switch(e.type){case"p":case",":let t=fe("link[rel=prev]");typeof t!="undefined"&<(t);break;case"n":case".":let r=fe("link[rel=next]");typeof r!="undefined"&<(r);break;case"Enter":let o=Ie();o instanceof HTMLLabelElement&&o.click()}});wi({viewport$:Oe,document$:ot});Ti({document$:ot,tablet$:hr});Si({document$:ot});Oi({viewport$:Oe,tablet$:hr});var rt=Kn(Se("header"),{viewport$:Oe}),jt=ot.pipe(m(()=>Se("main")),v(e=>Gn(e,{viewport$:Oe,header$:rt})),G(1)),xs=O(...ae("consent").map(e=>En(e,{target$:Ot})),...ae("dialog").map(e=>qn(e,{alert$:ro})),...ae("palette").map(e=>Jn(e)),...ae("progress").map(e=>Xn(e,{progress$:oo})),...ae("search").map(e=>ui(e,{index$:Ai,keyboard$:to})),...ae("source").map(e=>gi(e))),Es=C(()=>O(...ae("announce").map(e=>xn(e)),...ae("content").map(e=>Nn(e,{viewport$:Oe,target$:Ot,print$:_i})),...ae("content").map(e=>B("search.highlight")?di(e,{index$:Ai,location$:Ft}):S),...ae("header").map(e=>Yn(e,{viewport$:Oe,header$:rt,main$:jt})),...ae("header-title").map(e=>Bn(e,{viewport$:Oe,header$:rt})),...ae("sidebar").map(e=>e.getAttribute("data-md-type")==="navigation"?zr(Mi,()=>Zr(e,{viewport$:Oe,header$:rt,main$:jt})):zr(hr,()=>Zr(e,{viewport$:Oe,header$:rt,main$:jt}))),...ae("tabs").map(e=>yi(e,{viewport$:Oe,header$:rt})),...ae("toc").map(e=>xi(e,{viewport$:Oe,header$:rt,main$:jt,target$:Ot})),...ae("top").map(e=>Ei(e,{viewport$:Oe,header$:rt,main$:jt,target$:Ot})))),Ci=ot.pipe(v(()=>Es),Re(xs),G(1));Ci.subscribe();window.document$=ot;window.location$=Ft;window.target$=Ot;window.keyboard$=to;window.viewport$=Oe;window.tablet$=hr;window.screen$=Mi;window.print$=_i;window.alert$=ro;window.progress$=oo;window.component$=Ci;})(); +//# sourceMappingURL=bundle.f55a23d4.min.js.map + diff --git a/omniread/wiki/assets/javascripts/bundle.f55a23d4.min.js.map b/omniread/wiki/assets/javascripts/bundle.f55a23d4.min.js.map new file mode 100644 index 0000000..e3de73f --- /dev/null +++ b/omniread/wiki/assets/javascripts/bundle.f55a23d4.min.js.map @@ -0,0 +1,7 @@ +{ + "version": 3, + "sources": ["node_modules/focus-visible/dist/focus-visible.js", "node_modules/escape-html/index.js", "node_modules/clipboard/dist/clipboard.js", "src/templates/assets/javascripts/bundle.ts", "node_modules/tslib/tslib.es6.mjs", "node_modules/rxjs/src/internal/util/isFunction.ts", "node_modules/rxjs/src/internal/util/createErrorClass.ts", "node_modules/rxjs/src/internal/util/UnsubscriptionError.ts", "node_modules/rxjs/src/internal/util/arrRemove.ts", "node_modules/rxjs/src/internal/Subscription.ts", "node_modules/rxjs/src/internal/config.ts", "node_modules/rxjs/src/internal/scheduler/timeoutProvider.ts", "node_modules/rxjs/src/internal/util/reportUnhandledError.ts", "node_modules/rxjs/src/internal/util/noop.ts", "node_modules/rxjs/src/internal/NotificationFactories.ts", "node_modules/rxjs/src/internal/util/errorContext.ts", "node_modules/rxjs/src/internal/Subscriber.ts", "node_modules/rxjs/src/internal/symbol/observable.ts", "node_modules/rxjs/src/internal/util/identity.ts", "node_modules/rxjs/src/internal/util/pipe.ts", "node_modules/rxjs/src/internal/Observable.ts", "node_modules/rxjs/src/internal/util/lift.ts", "node_modules/rxjs/src/internal/operators/OperatorSubscriber.ts", "node_modules/rxjs/src/internal/scheduler/animationFrameProvider.ts", "node_modules/rxjs/src/internal/util/ObjectUnsubscribedError.ts", "node_modules/rxjs/src/internal/Subject.ts", "node_modules/rxjs/src/internal/BehaviorSubject.ts", "node_modules/rxjs/src/internal/scheduler/dateTimestampProvider.ts", "node_modules/rxjs/src/internal/ReplaySubject.ts", "node_modules/rxjs/src/internal/scheduler/Action.ts", "node_modules/rxjs/src/internal/scheduler/intervalProvider.ts", "node_modules/rxjs/src/internal/scheduler/AsyncAction.ts", "node_modules/rxjs/src/internal/Scheduler.ts", "node_modules/rxjs/src/internal/scheduler/AsyncScheduler.ts", "node_modules/rxjs/src/internal/scheduler/async.ts", "node_modules/rxjs/src/internal/scheduler/QueueAction.ts", "node_modules/rxjs/src/internal/scheduler/QueueScheduler.ts", "node_modules/rxjs/src/internal/scheduler/queue.ts", "node_modules/rxjs/src/internal/scheduler/AnimationFrameAction.ts", "node_modules/rxjs/src/internal/scheduler/AnimationFrameScheduler.ts", "node_modules/rxjs/src/internal/scheduler/animationFrame.ts", "node_modules/rxjs/src/internal/observable/empty.ts", "node_modules/rxjs/src/internal/util/isScheduler.ts", "node_modules/rxjs/src/internal/util/args.ts", "node_modules/rxjs/src/internal/util/isArrayLike.ts", "node_modules/rxjs/src/internal/util/isPromise.ts", "node_modules/rxjs/src/internal/util/isInteropObservable.ts", "node_modules/rxjs/src/internal/util/isAsyncIterable.ts", "node_modules/rxjs/src/internal/util/throwUnobservableError.ts", "node_modules/rxjs/src/internal/symbol/iterator.ts", "node_modules/rxjs/src/internal/util/isIterable.ts", "node_modules/rxjs/src/internal/util/isReadableStreamLike.ts", "node_modules/rxjs/src/internal/observable/innerFrom.ts", "node_modules/rxjs/src/internal/util/executeSchedule.ts", "node_modules/rxjs/src/internal/operators/observeOn.ts", "node_modules/rxjs/src/internal/operators/subscribeOn.ts", "node_modules/rxjs/src/internal/scheduled/scheduleObservable.ts", "node_modules/rxjs/src/internal/scheduled/schedulePromise.ts", "node_modules/rxjs/src/internal/scheduled/scheduleArray.ts", "node_modules/rxjs/src/internal/scheduled/scheduleIterable.ts", "node_modules/rxjs/src/internal/scheduled/scheduleAsyncIterable.ts", "node_modules/rxjs/src/internal/scheduled/scheduleReadableStreamLike.ts", "node_modules/rxjs/src/internal/scheduled/scheduled.ts", "node_modules/rxjs/src/internal/observable/from.ts", "node_modules/rxjs/src/internal/observable/of.ts", "node_modules/rxjs/src/internal/observable/throwError.ts", "node_modules/rxjs/src/internal/util/EmptyError.ts", "node_modules/rxjs/src/internal/util/isDate.ts", "node_modules/rxjs/src/internal/operators/map.ts", "node_modules/rxjs/src/internal/util/mapOneOrManyArgs.ts", "node_modules/rxjs/src/internal/util/argsArgArrayOrObject.ts", "node_modules/rxjs/src/internal/util/createObject.ts", "node_modules/rxjs/src/internal/observable/combineLatest.ts", "node_modules/rxjs/src/internal/operators/mergeInternals.ts", "node_modules/rxjs/src/internal/operators/mergeMap.ts", "node_modules/rxjs/src/internal/operators/mergeAll.ts", "node_modules/rxjs/src/internal/operators/concatAll.ts", "node_modules/rxjs/src/internal/observable/concat.ts", "node_modules/rxjs/src/internal/observable/defer.ts", "node_modules/rxjs/src/internal/observable/fromEvent.ts", "node_modules/rxjs/src/internal/observable/fromEventPattern.ts", "node_modules/rxjs/src/internal/observable/timer.ts", "node_modules/rxjs/src/internal/observable/merge.ts", "node_modules/rxjs/src/internal/observable/never.ts", "node_modules/rxjs/src/internal/util/argsOrArgArray.ts", "node_modules/rxjs/src/internal/operators/filter.ts", "node_modules/rxjs/src/internal/observable/zip.ts", "node_modules/rxjs/src/internal/operators/audit.ts", "node_modules/rxjs/src/internal/operators/auditTime.ts", "node_modules/rxjs/src/internal/operators/bufferCount.ts", "node_modules/rxjs/src/internal/operators/catchError.ts", "node_modules/rxjs/src/internal/operators/scanInternals.ts", "node_modules/rxjs/src/internal/operators/combineLatest.ts", "node_modules/rxjs/src/internal/operators/combineLatestWith.ts", "node_modules/rxjs/src/internal/operators/debounce.ts", "node_modules/rxjs/src/internal/operators/debounceTime.ts", "node_modules/rxjs/src/internal/operators/defaultIfEmpty.ts", "node_modules/rxjs/src/internal/operators/take.ts", "node_modules/rxjs/src/internal/operators/ignoreElements.ts", "node_modules/rxjs/src/internal/operators/mapTo.ts", "node_modules/rxjs/src/internal/operators/delayWhen.ts", "node_modules/rxjs/src/internal/operators/delay.ts", "node_modules/rxjs/src/internal/operators/distinctUntilChanged.ts", "node_modules/rxjs/src/internal/operators/distinctUntilKeyChanged.ts", "node_modules/rxjs/src/internal/operators/throwIfEmpty.ts", "node_modules/rxjs/src/internal/operators/endWith.ts", "node_modules/rxjs/src/internal/operators/finalize.ts", "node_modules/rxjs/src/internal/operators/first.ts", "node_modules/rxjs/src/internal/operators/takeLast.ts", "node_modules/rxjs/src/internal/operators/merge.ts", "node_modules/rxjs/src/internal/operators/mergeWith.ts", "node_modules/rxjs/src/internal/operators/repeat.ts", "node_modules/rxjs/src/internal/operators/scan.ts", "node_modules/rxjs/src/internal/operators/share.ts", "node_modules/rxjs/src/internal/operators/shareReplay.ts", "node_modules/rxjs/src/internal/operators/skip.ts", "node_modules/rxjs/src/internal/operators/skipUntil.ts", "node_modules/rxjs/src/internal/operators/startWith.ts", "node_modules/rxjs/src/internal/operators/switchMap.ts", "node_modules/rxjs/src/internal/operators/takeUntil.ts", "node_modules/rxjs/src/internal/operators/takeWhile.ts", "node_modules/rxjs/src/internal/operators/tap.ts", "node_modules/rxjs/src/internal/operators/throttle.ts", "node_modules/rxjs/src/internal/operators/throttleTime.ts", "node_modules/rxjs/src/internal/operators/withLatestFrom.ts", "node_modules/rxjs/src/internal/operators/zip.ts", "node_modules/rxjs/src/internal/operators/zipWith.ts", "src/templates/assets/javascripts/browser/document/index.ts", "src/templates/assets/javascripts/browser/element/_/index.ts", "src/templates/assets/javascripts/browser/element/focus/index.ts", "src/templates/assets/javascripts/browser/element/hover/index.ts", "src/templates/assets/javascripts/utilities/h/index.ts", "src/templates/assets/javascripts/utilities/round/index.ts", "src/templates/assets/javascripts/browser/script/index.ts", "src/templates/assets/javascripts/browser/element/size/_/index.ts", "src/templates/assets/javascripts/browser/element/size/content/index.ts", "src/templates/assets/javascripts/browser/element/offset/_/index.ts", "src/templates/assets/javascripts/browser/element/offset/content/index.ts", "src/templates/assets/javascripts/browser/element/visibility/index.ts", "src/templates/assets/javascripts/browser/toggle/index.ts", "src/templates/assets/javascripts/browser/keyboard/index.ts", "src/templates/assets/javascripts/browser/location/_/index.ts", "src/templates/assets/javascripts/browser/location/hash/index.ts", "src/templates/assets/javascripts/browser/media/index.ts", "src/templates/assets/javascripts/browser/request/index.ts", "src/templates/assets/javascripts/browser/viewport/offset/index.ts", "src/templates/assets/javascripts/browser/viewport/size/index.ts", "src/templates/assets/javascripts/browser/viewport/_/index.ts", "src/templates/assets/javascripts/browser/viewport/at/index.ts", "src/templates/assets/javascripts/browser/worker/index.ts", "src/templates/assets/javascripts/_/index.ts", "src/templates/assets/javascripts/components/_/index.ts", "src/templates/assets/javascripts/components/announce/index.ts", "src/templates/assets/javascripts/components/consent/index.ts", "src/templates/assets/javascripts/templates/tooltip/index.tsx", "src/templates/assets/javascripts/templates/annotation/index.tsx", "src/templates/assets/javascripts/templates/clipboard/index.tsx", "src/templates/assets/javascripts/templates/search/index.tsx", "src/templates/assets/javascripts/templates/source/index.tsx", "src/templates/assets/javascripts/templates/tabbed/index.tsx", "src/templates/assets/javascripts/templates/table/index.tsx", "src/templates/assets/javascripts/templates/version/index.tsx", "src/templates/assets/javascripts/components/tooltip2/index.ts", "src/templates/assets/javascripts/components/content/annotation/_/index.ts", "src/templates/assets/javascripts/components/content/annotation/list/index.ts", "src/templates/assets/javascripts/components/content/annotation/block/index.ts", "src/templates/assets/javascripts/components/content/code/_/index.ts", "src/templates/assets/javascripts/components/content/details/index.ts", "src/templates/assets/javascripts/components/content/mermaid/index.css", "src/templates/assets/javascripts/components/content/mermaid/index.ts", "src/templates/assets/javascripts/components/content/table/index.ts", "src/templates/assets/javascripts/components/content/tabs/index.ts", "src/templates/assets/javascripts/components/content/_/index.ts", "src/templates/assets/javascripts/components/dialog/index.ts", "src/templates/assets/javascripts/components/tooltip/index.ts", "src/templates/assets/javascripts/components/header/_/index.ts", "src/templates/assets/javascripts/components/header/title/index.ts", "src/templates/assets/javascripts/components/main/index.ts", "src/templates/assets/javascripts/components/palette/index.ts", "src/templates/assets/javascripts/components/progress/index.ts", "src/templates/assets/javascripts/integrations/clipboard/index.ts", "src/templates/assets/javascripts/integrations/sitemap/index.ts", "src/templates/assets/javascripts/integrations/instant/index.ts", "src/templates/assets/javascripts/integrations/search/highlighter/index.ts", "src/templates/assets/javascripts/integrations/search/worker/message/index.ts", "src/templates/assets/javascripts/integrations/search/worker/_/index.ts", "src/templates/assets/javascripts/integrations/version/findurl/index.ts", "src/templates/assets/javascripts/integrations/version/index.ts", "src/templates/assets/javascripts/components/search/query/index.ts", "src/templates/assets/javascripts/components/search/result/index.ts", "src/templates/assets/javascripts/components/search/share/index.ts", "src/templates/assets/javascripts/components/search/suggest/index.ts", "src/templates/assets/javascripts/components/search/_/index.ts", "src/templates/assets/javascripts/components/search/highlight/index.ts", "src/templates/assets/javascripts/components/sidebar/index.ts", "src/templates/assets/javascripts/components/source/facts/github/index.ts", "src/templates/assets/javascripts/components/source/facts/gitlab/index.ts", "src/templates/assets/javascripts/components/source/facts/_/index.ts", "src/templates/assets/javascripts/components/source/_/index.ts", "src/templates/assets/javascripts/components/tabs/index.ts", "src/templates/assets/javascripts/components/toc/index.ts", "src/templates/assets/javascripts/components/top/index.ts", "src/templates/assets/javascripts/patches/ellipsis/index.ts", "src/templates/assets/javascripts/patches/indeterminate/index.ts", "src/templates/assets/javascripts/patches/scrollfix/index.ts", "src/templates/assets/javascripts/patches/scrolllock/index.ts", "src/templates/assets/javascripts/polyfills/index.ts"], + "sourcesContent": ["(function (global, factory) {\n typeof exports === 'object' && typeof module !== 'undefined' ? factory() :\n typeof define === 'function' && define.amd ? define(factory) :\n (factory());\n}(this, (function () { 'use strict';\n\n /**\n * Applies the :focus-visible polyfill at the given scope.\n * A scope in this case is either the top-level Document or a Shadow Root.\n *\n * @param {(Document|ShadowRoot)} scope\n * @see https://github.com/WICG/focus-visible\n */\n function applyFocusVisiblePolyfill(scope) {\n var hadKeyboardEvent = true;\n var hadFocusVisibleRecently = false;\n var hadFocusVisibleRecentlyTimeout = null;\n\n var inputTypesAllowlist = {\n text: true,\n search: true,\n url: true,\n tel: true,\n email: true,\n password: true,\n number: true,\n date: true,\n month: true,\n week: true,\n time: true,\n datetime: true,\n 'datetime-local': true\n };\n\n /**\n * Helper function for legacy browsers and iframes which sometimes focus\n * elements like document, body, and non-interactive SVG.\n * @param {Element} el\n */\n function isValidFocusTarget(el) {\n if (\n el &&\n el !== document &&\n el.nodeName !== 'HTML' &&\n el.nodeName !== 'BODY' &&\n 'classList' in el &&\n 'contains' in el.classList\n ) {\n return true;\n }\n return false;\n }\n\n /**\n * Computes whether the given element should automatically trigger the\n * `focus-visible` class being added, i.e. whether it should always match\n * `:focus-visible` when focused.\n * @param {Element} el\n * @return {boolean}\n */\n function focusTriggersKeyboardModality(el) {\n var type = el.type;\n var tagName = el.tagName;\n\n if (tagName === 'INPUT' && inputTypesAllowlist[type] && !el.readOnly) {\n return true;\n }\n\n if (tagName === 'TEXTAREA' && !el.readOnly) {\n return true;\n }\n\n if (el.isContentEditable) {\n return true;\n }\n\n return false;\n }\n\n /**\n * Add the `focus-visible` class to the given element if it was not added by\n * the author.\n * @param {Element} el\n */\n function addFocusVisibleClass(el) {\n if (el.classList.contains('focus-visible')) {\n return;\n }\n el.classList.add('focus-visible');\n el.setAttribute('data-focus-visible-added', '');\n }\n\n /**\n * Remove the `focus-visible` class from the given element if it was not\n * originally added by the author.\n * @param {Element} el\n */\n function removeFocusVisibleClass(el) {\n if (!el.hasAttribute('data-focus-visible-added')) {\n return;\n }\n el.classList.remove('focus-visible');\n el.removeAttribute('data-focus-visible-added');\n }\n\n /**\n * If the most recent user interaction was via the keyboard;\n * and the key press did not include a meta, alt/option, or control key;\n * then the modality is keyboard. Otherwise, the modality is not keyboard.\n * Apply `focus-visible` to any current active element and keep track\n * of our keyboard modality state with `hadKeyboardEvent`.\n * @param {KeyboardEvent} e\n */\n function onKeyDown(e) {\n if (e.metaKey || e.altKey || e.ctrlKey) {\n return;\n }\n\n if (isValidFocusTarget(scope.activeElement)) {\n addFocusVisibleClass(scope.activeElement);\n }\n\n hadKeyboardEvent = true;\n }\n\n /**\n * If at any point a user clicks with a pointing device, ensure that we change\n * the modality away from keyboard.\n * This avoids the situation where a user presses a key on an already focused\n * element, and then clicks on a different element, focusing it with a\n * pointing device, while we still think we're in keyboard modality.\n * @param {Event} e\n */\n function onPointerDown(e) {\n hadKeyboardEvent = false;\n }\n\n /**\n * On `focus`, add the `focus-visible` class to the target if:\n * - the target received focus as a result of keyboard navigation, or\n * - the event target is an element that will likely require interaction\n * via the keyboard (e.g. a text box)\n * @param {Event} e\n */\n function onFocus(e) {\n // Prevent IE from focusing the document or HTML element.\n if (!isValidFocusTarget(e.target)) {\n return;\n }\n\n if (hadKeyboardEvent || focusTriggersKeyboardModality(e.target)) {\n addFocusVisibleClass(e.target);\n }\n }\n\n /**\n * On `blur`, remove the `focus-visible` class from the target.\n * @param {Event} e\n */\n function onBlur(e) {\n if (!isValidFocusTarget(e.target)) {\n return;\n }\n\n if (\n e.target.classList.contains('focus-visible') ||\n e.target.hasAttribute('data-focus-visible-added')\n ) {\n // To detect a tab/window switch, we look for a blur event followed\n // rapidly by a visibility change.\n // If we don't see a visibility change within 100ms, it's probably a\n // regular focus change.\n hadFocusVisibleRecently = true;\n window.clearTimeout(hadFocusVisibleRecentlyTimeout);\n hadFocusVisibleRecentlyTimeout = window.setTimeout(function() {\n hadFocusVisibleRecently = false;\n }, 100);\n removeFocusVisibleClass(e.target);\n }\n }\n\n /**\n * If the user changes tabs, keep track of whether or not the previously\n * focused element had .focus-visible.\n * @param {Event} e\n */\n function onVisibilityChange(e) {\n if (document.visibilityState === 'hidden') {\n // If the tab becomes active again, the browser will handle calling focus\n // on the element (Safari actually calls it twice).\n // If this tab change caused a blur on an element with focus-visible,\n // re-apply the class when the user switches back to the tab.\n if (hadFocusVisibleRecently) {\n hadKeyboardEvent = true;\n }\n addInitialPointerMoveListeners();\n }\n }\n\n /**\n * Add a group of listeners to detect usage of any pointing devices.\n * These listeners will be added when the polyfill first loads, and anytime\n * the window is blurred, so that they are active when the window regains\n * focus.\n */\n function addInitialPointerMoveListeners() {\n document.addEventListener('mousemove', onInitialPointerMove);\n document.addEventListener('mousedown', onInitialPointerMove);\n document.addEventListener('mouseup', onInitialPointerMove);\n document.addEventListener('pointermove', onInitialPointerMove);\n document.addEventListener('pointerdown', onInitialPointerMove);\n document.addEventListener('pointerup', onInitialPointerMove);\n document.addEventListener('touchmove', onInitialPointerMove);\n document.addEventListener('touchstart', onInitialPointerMove);\n document.addEventListener('touchend', onInitialPointerMove);\n }\n\n function removeInitialPointerMoveListeners() {\n document.removeEventListener('mousemove', onInitialPointerMove);\n document.removeEventListener('mousedown', onInitialPointerMove);\n document.removeEventListener('mouseup', onInitialPointerMove);\n document.removeEventListener('pointermove', onInitialPointerMove);\n document.removeEventListener('pointerdown', onInitialPointerMove);\n document.removeEventListener('pointerup', onInitialPointerMove);\n document.removeEventListener('touchmove', onInitialPointerMove);\n document.removeEventListener('touchstart', onInitialPointerMove);\n document.removeEventListener('touchend', onInitialPointerMove);\n }\n\n /**\n * When the polfyill first loads, assume the user is in keyboard modality.\n * If any event is received from a pointing device (e.g. mouse, pointer,\n * touch), turn off keyboard modality.\n * This accounts for situations where focus enters the page from the URL bar.\n * @param {Event} e\n */\n function onInitialPointerMove(e) {\n // Work around a Safari quirk that fires a mousemove on whenever the\n // window blurs, even if you're tabbing out of the page. \u00AF\\_(\u30C4)_/\u00AF\n if (e.target.nodeName && e.target.nodeName.toLowerCase() === 'html') {\n return;\n }\n\n hadKeyboardEvent = false;\n removeInitialPointerMoveListeners();\n }\n\n // For some kinds of state, we are interested in changes at the global scope\n // only. For example, global pointer input, global key presses and global\n // visibility change should affect the state at every scope:\n document.addEventListener('keydown', onKeyDown, true);\n document.addEventListener('mousedown', onPointerDown, true);\n document.addEventListener('pointerdown', onPointerDown, true);\n document.addEventListener('touchstart', onPointerDown, true);\n document.addEventListener('visibilitychange', onVisibilityChange, true);\n\n addInitialPointerMoveListeners();\n\n // For focus and blur, we specifically care about state changes in the local\n // scope. This is because focus / blur events that originate from within a\n // shadow root are not re-dispatched from the host element if it was already\n // the active element in its own scope:\n scope.addEventListener('focus', onFocus, true);\n scope.addEventListener('blur', onBlur, true);\n\n // We detect that a node is a ShadowRoot by ensuring that it is a\n // DocumentFragment and also has a host property. This check covers native\n // implementation and polyfill implementation transparently. If we only cared\n // about the native implementation, we could just check if the scope was\n // an instance of a ShadowRoot.\n if (scope.nodeType === Node.DOCUMENT_FRAGMENT_NODE && scope.host) {\n // Since a ShadowRoot is a special kind of DocumentFragment, it does not\n // have a root element to add a class to. So, we add this attribute to the\n // host element instead:\n scope.host.setAttribute('data-js-focus-visible', '');\n } else if (scope.nodeType === Node.DOCUMENT_NODE) {\n document.documentElement.classList.add('js-focus-visible');\n document.documentElement.setAttribute('data-js-focus-visible', '');\n }\n }\n\n // It is important to wrap all references to global window and document in\n // these checks to support server-side rendering use cases\n // @see https://github.com/WICG/focus-visible/issues/199\n if (typeof window !== 'undefined' && typeof document !== 'undefined') {\n // Make the polyfill helper globally available. This can be used as a signal\n // to interested libraries that wish to coordinate with the polyfill for e.g.,\n // applying the polyfill to a shadow root:\n window.applyFocusVisiblePolyfill = applyFocusVisiblePolyfill;\n\n // Notify interested libraries of the polyfill's presence, in case the\n // polyfill was loaded lazily:\n var event;\n\n try {\n event = new CustomEvent('focus-visible-polyfill-ready');\n } catch (error) {\n // IE11 does not support using CustomEvent as a constructor directly:\n event = document.createEvent('CustomEvent');\n event.initCustomEvent('focus-visible-polyfill-ready', false, false, {});\n }\n\n window.dispatchEvent(event);\n }\n\n if (typeof document !== 'undefined') {\n // Apply the polyfill to the global document, so that no JavaScript\n // coordination is required to use the polyfill in the top-level document:\n applyFocusVisiblePolyfill(document);\n }\n\n})));\n", "/*!\n * escape-html\n * Copyright(c) 2012-2013 TJ Holowaychuk\n * Copyright(c) 2015 Andreas Lubbe\n * Copyright(c) 2015 Tiancheng \"Timothy\" Gu\n * MIT Licensed\n */\n\n'use strict';\n\n/**\n * Module variables.\n * @private\n */\n\nvar matchHtmlRegExp = /[\"'&<>]/;\n\n/**\n * Module exports.\n * @public\n */\n\nmodule.exports = escapeHtml;\n\n/**\n * Escape special characters in the given string of html.\n *\n * @param {string} string The string to escape for inserting into HTML\n * @return {string}\n * @public\n */\n\nfunction escapeHtml(string) {\n var str = '' + string;\n var match = matchHtmlRegExp.exec(str);\n\n if (!match) {\n return str;\n }\n\n var escape;\n var html = '';\n var index = 0;\n var lastIndex = 0;\n\n for (index = match.index; index < str.length; index++) {\n switch (str.charCodeAt(index)) {\n case 34: // \"\n escape = '"';\n break;\n case 38: // &\n escape = '&';\n break;\n case 39: // '\n escape = ''';\n break;\n case 60: // <\n escape = '<';\n break;\n case 62: // >\n escape = '>';\n break;\n default:\n continue;\n }\n\n if (lastIndex !== index) {\n html += str.substring(lastIndex, index);\n }\n\n lastIndex = index + 1;\n html += escape;\n }\n\n return lastIndex !== index\n ? html + str.substring(lastIndex, index)\n : html;\n}\n", "/*!\n * clipboard.js v2.0.11\n * https://clipboardjs.com/\n *\n * Licensed MIT \u00A9 Zeno Rocha\n */\n(function webpackUniversalModuleDefinition(root, factory) {\n\tif(typeof exports === 'object' && typeof module === 'object')\n\t\tmodule.exports = factory();\n\telse if(typeof define === 'function' && define.amd)\n\t\tdefine([], factory);\n\telse if(typeof exports === 'object')\n\t\texports[\"ClipboardJS\"] = factory();\n\telse\n\t\troot[\"ClipboardJS\"] = factory();\n})(this, function() {\nreturn /******/ (function() { // webpackBootstrap\n/******/ \tvar __webpack_modules__ = ({\n\n/***/ 686:\n/***/ (function(__unused_webpack_module, __webpack_exports__, __webpack_require__) {\n\n\"use strict\";\n\n// EXPORTS\n__webpack_require__.d(__webpack_exports__, {\n \"default\": function() { return /* binding */ clipboard; }\n});\n\n// EXTERNAL MODULE: ./node_modules/tiny-emitter/index.js\nvar tiny_emitter = __webpack_require__(279);\nvar tiny_emitter_default = /*#__PURE__*/__webpack_require__.n(tiny_emitter);\n// EXTERNAL MODULE: ./node_modules/good-listener/src/listen.js\nvar listen = __webpack_require__(370);\nvar listen_default = /*#__PURE__*/__webpack_require__.n(listen);\n// EXTERNAL MODULE: ./node_modules/select/src/select.js\nvar src_select = __webpack_require__(817);\nvar select_default = /*#__PURE__*/__webpack_require__.n(src_select);\n;// CONCATENATED MODULE: ./src/common/command.js\n/**\n * Executes a given operation type.\n * @param {String} type\n * @return {Boolean}\n */\nfunction command(type) {\n try {\n return document.execCommand(type);\n } catch (err) {\n return false;\n }\n}\n;// CONCATENATED MODULE: ./src/actions/cut.js\n\n\n/**\n * Cut action wrapper.\n * @param {String|HTMLElement} target\n * @return {String}\n */\n\nvar ClipboardActionCut = function ClipboardActionCut(target) {\n var selectedText = select_default()(target);\n command('cut');\n return selectedText;\n};\n\n/* harmony default export */ var actions_cut = (ClipboardActionCut);\n;// CONCATENATED MODULE: ./src/common/create-fake-element.js\n/**\n * Creates a fake textarea element with a value.\n * @param {String} value\n * @return {HTMLElement}\n */\nfunction createFakeElement(value) {\n var isRTL = document.documentElement.getAttribute('dir') === 'rtl';\n var fakeElement = document.createElement('textarea'); // Prevent zooming on iOS\n\n fakeElement.style.fontSize = '12pt'; // Reset box model\n\n fakeElement.style.border = '0';\n fakeElement.style.padding = '0';\n fakeElement.style.margin = '0'; // Move element out of screen horizontally\n\n fakeElement.style.position = 'absolute';\n fakeElement.style[isRTL ? 'right' : 'left'] = '-9999px'; // Move element to the same position vertically\n\n var yPosition = window.pageYOffset || document.documentElement.scrollTop;\n fakeElement.style.top = \"\".concat(yPosition, \"px\");\n fakeElement.setAttribute('readonly', '');\n fakeElement.value = value;\n return fakeElement;\n}\n;// CONCATENATED MODULE: ./src/actions/copy.js\n\n\n\n/**\n * Create fake copy action wrapper using a fake element.\n * @param {String} target\n * @param {Object} options\n * @return {String}\n */\n\nvar fakeCopyAction = function fakeCopyAction(value, options) {\n var fakeElement = createFakeElement(value);\n options.container.appendChild(fakeElement);\n var selectedText = select_default()(fakeElement);\n command('copy');\n fakeElement.remove();\n return selectedText;\n};\n/**\n * Copy action wrapper.\n * @param {String|HTMLElement} target\n * @param {Object} options\n * @return {String}\n */\n\n\nvar ClipboardActionCopy = function ClipboardActionCopy(target) {\n var options = arguments.length > 1 && arguments[1] !== undefined ? arguments[1] : {\n container: document.body\n };\n var selectedText = '';\n\n if (typeof target === 'string') {\n selectedText = fakeCopyAction(target, options);\n } else if (target instanceof HTMLInputElement && !['text', 'search', 'url', 'tel', 'password'].includes(target === null || target === void 0 ? void 0 : target.type)) {\n // If input type doesn't support `setSelectionRange`. Simulate it. https://developer.mozilla.org/en-US/docs/Web/API/HTMLInputElement/setSelectionRange\n selectedText = fakeCopyAction(target.value, options);\n } else {\n selectedText = select_default()(target);\n command('copy');\n }\n\n return selectedText;\n};\n\n/* harmony default export */ var actions_copy = (ClipboardActionCopy);\n;// CONCATENATED MODULE: ./src/actions/default.js\nfunction _typeof(obj) { \"@babel/helpers - typeof\"; if (typeof Symbol === \"function\" && typeof Symbol.iterator === \"symbol\") { _typeof = function _typeof(obj) { return typeof obj; }; } else { _typeof = function _typeof(obj) { return obj && typeof Symbol === \"function\" && obj.constructor === Symbol && obj !== Symbol.prototype ? \"symbol\" : typeof obj; }; } return _typeof(obj); }\n\n\n\n/**\n * Inner function which performs selection from either `text` or `target`\n * properties and then executes copy or cut operations.\n * @param {Object} options\n */\n\nvar ClipboardActionDefault = function ClipboardActionDefault() {\n var options = arguments.length > 0 && arguments[0] !== undefined ? arguments[0] : {};\n // Defines base properties passed from constructor.\n var _options$action = options.action,\n action = _options$action === void 0 ? 'copy' : _options$action,\n container = options.container,\n target = options.target,\n text = options.text; // Sets the `action` to be performed which can be either 'copy' or 'cut'.\n\n if (action !== 'copy' && action !== 'cut') {\n throw new Error('Invalid \"action\" value, use either \"copy\" or \"cut\"');\n } // Sets the `target` property using an element that will be have its content copied.\n\n\n if (target !== undefined) {\n if (target && _typeof(target) === 'object' && target.nodeType === 1) {\n if (action === 'copy' && target.hasAttribute('disabled')) {\n throw new Error('Invalid \"target\" attribute. Please use \"readonly\" instead of \"disabled\" attribute');\n }\n\n if (action === 'cut' && (target.hasAttribute('readonly') || target.hasAttribute('disabled'))) {\n throw new Error('Invalid \"target\" attribute. You can\\'t cut text from elements with \"readonly\" or \"disabled\" attributes');\n }\n } else {\n throw new Error('Invalid \"target\" value, use a valid Element');\n }\n } // Define selection strategy based on `text` property.\n\n\n if (text) {\n return actions_copy(text, {\n container: container\n });\n } // Defines which selection strategy based on `target` property.\n\n\n if (target) {\n return action === 'cut' ? actions_cut(target) : actions_copy(target, {\n container: container\n });\n }\n};\n\n/* harmony default export */ var actions_default = (ClipboardActionDefault);\n;// CONCATENATED MODULE: ./src/clipboard.js\nfunction clipboard_typeof(obj) { \"@babel/helpers - typeof\"; if (typeof Symbol === \"function\" && typeof Symbol.iterator === \"symbol\") { clipboard_typeof = function _typeof(obj) { return typeof obj; }; } else { clipboard_typeof = function _typeof(obj) { return obj && typeof Symbol === \"function\" && obj.constructor === Symbol && obj !== Symbol.prototype ? \"symbol\" : typeof obj; }; } return clipboard_typeof(obj); }\n\nfunction _classCallCheck(instance, Constructor) { if (!(instance instanceof Constructor)) { throw new TypeError(\"Cannot call a class as a function\"); } }\n\nfunction _defineProperties(target, props) { for (var i = 0; i < props.length; i++) { var descriptor = props[i]; descriptor.enumerable = descriptor.enumerable || false; descriptor.configurable = true; if (\"value\" in descriptor) descriptor.writable = true; Object.defineProperty(target, descriptor.key, descriptor); } }\n\nfunction _createClass(Constructor, protoProps, staticProps) { if (protoProps) _defineProperties(Constructor.prototype, protoProps); if (staticProps) _defineProperties(Constructor, staticProps); return Constructor; }\n\nfunction _inherits(subClass, superClass) { if (typeof superClass !== \"function\" && superClass !== null) { throw new TypeError(\"Super expression must either be null or a function\"); } subClass.prototype = Object.create(superClass && superClass.prototype, { constructor: { value: subClass, writable: true, configurable: true } }); if (superClass) _setPrototypeOf(subClass, superClass); }\n\nfunction _setPrototypeOf(o, p) { _setPrototypeOf = Object.setPrototypeOf || function _setPrototypeOf(o, p) { o.__proto__ = p; return o; }; return _setPrototypeOf(o, p); }\n\nfunction _createSuper(Derived) { var hasNativeReflectConstruct = _isNativeReflectConstruct(); return function _createSuperInternal() { var Super = _getPrototypeOf(Derived), result; if (hasNativeReflectConstruct) { var NewTarget = _getPrototypeOf(this).constructor; result = Reflect.construct(Super, arguments, NewTarget); } else { result = Super.apply(this, arguments); } return _possibleConstructorReturn(this, result); }; }\n\nfunction _possibleConstructorReturn(self, call) { if (call && (clipboard_typeof(call) === \"object\" || typeof call === \"function\")) { return call; } return _assertThisInitialized(self); }\n\nfunction _assertThisInitialized(self) { if (self === void 0) { throw new ReferenceError(\"this hasn't been initialised - super() hasn't been called\"); } return self; }\n\nfunction _isNativeReflectConstruct() { if (typeof Reflect === \"undefined\" || !Reflect.construct) return false; if (Reflect.construct.sham) return false; if (typeof Proxy === \"function\") return true; try { Date.prototype.toString.call(Reflect.construct(Date, [], function () {})); return true; } catch (e) { return false; } }\n\nfunction _getPrototypeOf(o) { _getPrototypeOf = Object.setPrototypeOf ? Object.getPrototypeOf : function _getPrototypeOf(o) { return o.__proto__ || Object.getPrototypeOf(o); }; return _getPrototypeOf(o); }\n\n\n\n\n\n\n/**\n * Helper function to retrieve attribute value.\n * @param {String} suffix\n * @param {Element} element\n */\n\nfunction getAttributeValue(suffix, element) {\n var attribute = \"data-clipboard-\".concat(suffix);\n\n if (!element.hasAttribute(attribute)) {\n return;\n }\n\n return element.getAttribute(attribute);\n}\n/**\n * Base class which takes one or more elements, adds event listeners to them,\n * and instantiates a new `ClipboardAction` on each click.\n */\n\n\nvar Clipboard = /*#__PURE__*/function (_Emitter) {\n _inherits(Clipboard, _Emitter);\n\n var _super = _createSuper(Clipboard);\n\n /**\n * @param {String|HTMLElement|HTMLCollection|NodeList} trigger\n * @param {Object} options\n */\n function Clipboard(trigger, options) {\n var _this;\n\n _classCallCheck(this, Clipboard);\n\n _this = _super.call(this);\n\n _this.resolveOptions(options);\n\n _this.listenClick(trigger);\n\n return _this;\n }\n /**\n * Defines if attributes would be resolved using internal setter functions\n * or custom functions that were passed in the constructor.\n * @param {Object} options\n */\n\n\n _createClass(Clipboard, [{\n key: \"resolveOptions\",\n value: function resolveOptions() {\n var options = arguments.length > 0 && arguments[0] !== undefined ? arguments[0] : {};\n this.action = typeof options.action === 'function' ? options.action : this.defaultAction;\n this.target = typeof options.target === 'function' ? options.target : this.defaultTarget;\n this.text = typeof options.text === 'function' ? options.text : this.defaultText;\n this.container = clipboard_typeof(options.container) === 'object' ? options.container : document.body;\n }\n /**\n * Adds a click event listener to the passed trigger.\n * @param {String|HTMLElement|HTMLCollection|NodeList} trigger\n */\n\n }, {\n key: \"listenClick\",\n value: function listenClick(trigger) {\n var _this2 = this;\n\n this.listener = listen_default()(trigger, 'click', function (e) {\n return _this2.onClick(e);\n });\n }\n /**\n * Defines a new `ClipboardAction` on each click event.\n * @param {Event} e\n */\n\n }, {\n key: \"onClick\",\n value: function onClick(e) {\n var trigger = e.delegateTarget || e.currentTarget;\n var action = this.action(trigger) || 'copy';\n var text = actions_default({\n action: action,\n container: this.container,\n target: this.target(trigger),\n text: this.text(trigger)\n }); // Fires an event based on the copy operation result.\n\n this.emit(text ? 'success' : 'error', {\n action: action,\n text: text,\n trigger: trigger,\n clearSelection: function clearSelection() {\n if (trigger) {\n trigger.focus();\n }\n\n window.getSelection().removeAllRanges();\n }\n });\n }\n /**\n * Default `action` lookup function.\n * @param {Element} trigger\n */\n\n }, {\n key: \"defaultAction\",\n value: function defaultAction(trigger) {\n return getAttributeValue('action', trigger);\n }\n /**\n * Default `target` lookup function.\n * @param {Element} trigger\n */\n\n }, {\n key: \"defaultTarget\",\n value: function defaultTarget(trigger) {\n var selector = getAttributeValue('target', trigger);\n\n if (selector) {\n return document.querySelector(selector);\n }\n }\n /**\n * Allow fire programmatically a copy action\n * @param {String|HTMLElement} target\n * @param {Object} options\n * @returns Text copied.\n */\n\n }, {\n key: \"defaultText\",\n\n /**\n * Default `text` lookup function.\n * @param {Element} trigger\n */\n value: function defaultText(trigger) {\n return getAttributeValue('text', trigger);\n }\n /**\n * Destroy lifecycle.\n */\n\n }, {\n key: \"destroy\",\n value: function destroy() {\n this.listener.destroy();\n }\n }], [{\n key: \"copy\",\n value: function copy(target) {\n var options = arguments.length > 1 && arguments[1] !== undefined ? arguments[1] : {\n container: document.body\n };\n return actions_copy(target, options);\n }\n /**\n * Allow fire programmatically a cut action\n * @param {String|HTMLElement} target\n * @returns Text cutted.\n */\n\n }, {\n key: \"cut\",\n value: function cut(target) {\n return actions_cut(target);\n }\n /**\n * Returns the support of the given action, or all actions if no action is\n * given.\n * @param {String} [action]\n */\n\n }, {\n key: \"isSupported\",\n value: function isSupported() {\n var action = arguments.length > 0 && arguments[0] !== undefined ? arguments[0] : ['copy', 'cut'];\n var actions = typeof action === 'string' ? [action] : action;\n var support = !!document.queryCommandSupported;\n actions.forEach(function (action) {\n support = support && !!document.queryCommandSupported(action);\n });\n return support;\n }\n }]);\n\n return Clipboard;\n}((tiny_emitter_default()));\n\n/* harmony default export */ var clipboard = (Clipboard);\n\n/***/ }),\n\n/***/ 828:\n/***/ (function(module) {\n\nvar DOCUMENT_NODE_TYPE = 9;\n\n/**\n * A polyfill for Element.matches()\n */\nif (typeof Element !== 'undefined' && !Element.prototype.matches) {\n var proto = Element.prototype;\n\n proto.matches = proto.matchesSelector ||\n proto.mozMatchesSelector ||\n proto.msMatchesSelector ||\n proto.oMatchesSelector ||\n proto.webkitMatchesSelector;\n}\n\n/**\n * Finds the closest parent that matches a selector.\n *\n * @param {Element} element\n * @param {String} selector\n * @return {Function}\n */\nfunction closest (element, selector) {\n while (element && element.nodeType !== DOCUMENT_NODE_TYPE) {\n if (typeof element.matches === 'function' &&\n element.matches(selector)) {\n return element;\n }\n element = element.parentNode;\n }\n}\n\nmodule.exports = closest;\n\n\n/***/ }),\n\n/***/ 438:\n/***/ (function(module, __unused_webpack_exports, __webpack_require__) {\n\nvar closest = __webpack_require__(828);\n\n/**\n * Delegates event to a selector.\n *\n * @param {Element} element\n * @param {String} selector\n * @param {String} type\n * @param {Function} callback\n * @param {Boolean} useCapture\n * @return {Object}\n */\nfunction _delegate(element, selector, type, callback, useCapture) {\n var listenerFn = listener.apply(this, arguments);\n\n element.addEventListener(type, listenerFn, useCapture);\n\n return {\n destroy: function() {\n element.removeEventListener(type, listenerFn, useCapture);\n }\n }\n}\n\n/**\n * Delegates event to a selector.\n *\n * @param {Element|String|Array} [elements]\n * @param {String} selector\n * @param {String} type\n * @param {Function} callback\n * @param {Boolean} useCapture\n * @return {Object}\n */\nfunction delegate(elements, selector, type, callback, useCapture) {\n // Handle the regular Element usage\n if (typeof elements.addEventListener === 'function') {\n return _delegate.apply(null, arguments);\n }\n\n // Handle Element-less usage, it defaults to global delegation\n if (typeof type === 'function') {\n // Use `document` as the first parameter, then apply arguments\n // This is a short way to .unshift `arguments` without running into deoptimizations\n return _delegate.bind(null, document).apply(null, arguments);\n }\n\n // Handle Selector-based usage\n if (typeof elements === 'string') {\n elements = document.querySelectorAll(elements);\n }\n\n // Handle Array-like based usage\n return Array.prototype.map.call(elements, function (element) {\n return _delegate(element, selector, type, callback, useCapture);\n });\n}\n\n/**\n * Finds closest match and invokes callback.\n *\n * @param {Element} element\n * @param {String} selector\n * @param {String} type\n * @param {Function} callback\n * @return {Function}\n */\nfunction listener(element, selector, type, callback) {\n return function(e) {\n e.delegateTarget = closest(e.target, selector);\n\n if (e.delegateTarget) {\n callback.call(element, e);\n }\n }\n}\n\nmodule.exports = delegate;\n\n\n/***/ }),\n\n/***/ 879:\n/***/ (function(__unused_webpack_module, exports) {\n\n/**\n * Check if argument is a HTML element.\n *\n * @param {Object} value\n * @return {Boolean}\n */\nexports.node = function(value) {\n return value !== undefined\n && value instanceof HTMLElement\n && value.nodeType === 1;\n};\n\n/**\n * Check if argument is a list of HTML elements.\n *\n * @param {Object} value\n * @return {Boolean}\n */\nexports.nodeList = function(value) {\n var type = Object.prototype.toString.call(value);\n\n return value !== undefined\n && (type === '[object NodeList]' || type === '[object HTMLCollection]')\n && ('length' in value)\n && (value.length === 0 || exports.node(value[0]));\n};\n\n/**\n * Check if argument is a string.\n *\n * @param {Object} value\n * @return {Boolean}\n */\nexports.string = function(value) {\n return typeof value === 'string'\n || value instanceof String;\n};\n\n/**\n * Check if argument is a function.\n *\n * @param {Object} value\n * @return {Boolean}\n */\nexports.fn = function(value) {\n var type = Object.prototype.toString.call(value);\n\n return type === '[object Function]';\n};\n\n\n/***/ }),\n\n/***/ 370:\n/***/ (function(module, __unused_webpack_exports, __webpack_require__) {\n\nvar is = __webpack_require__(879);\nvar delegate = __webpack_require__(438);\n\n/**\n * Validates all params and calls the right\n * listener function based on its target type.\n *\n * @param {String|HTMLElement|HTMLCollection|NodeList} target\n * @param {String} type\n * @param {Function} callback\n * @return {Object}\n */\nfunction listen(target, type, callback) {\n if (!target && !type && !callback) {\n throw new Error('Missing required arguments');\n }\n\n if (!is.string(type)) {\n throw new TypeError('Second argument must be a String');\n }\n\n if (!is.fn(callback)) {\n throw new TypeError('Third argument must be a Function');\n }\n\n if (is.node(target)) {\n return listenNode(target, type, callback);\n }\n else if (is.nodeList(target)) {\n return listenNodeList(target, type, callback);\n }\n else if (is.string(target)) {\n return listenSelector(target, type, callback);\n }\n else {\n throw new TypeError('First argument must be a String, HTMLElement, HTMLCollection, or NodeList');\n }\n}\n\n/**\n * Adds an event listener to a HTML element\n * and returns a remove listener function.\n *\n * @param {HTMLElement} node\n * @param {String} type\n * @param {Function} callback\n * @return {Object}\n */\nfunction listenNode(node, type, callback) {\n node.addEventListener(type, callback);\n\n return {\n destroy: function() {\n node.removeEventListener(type, callback);\n }\n }\n}\n\n/**\n * Add an event listener to a list of HTML elements\n * and returns a remove listener function.\n *\n * @param {NodeList|HTMLCollection} nodeList\n * @param {String} type\n * @param {Function} callback\n * @return {Object}\n */\nfunction listenNodeList(nodeList, type, callback) {\n Array.prototype.forEach.call(nodeList, function(node) {\n node.addEventListener(type, callback);\n });\n\n return {\n destroy: function() {\n Array.prototype.forEach.call(nodeList, function(node) {\n node.removeEventListener(type, callback);\n });\n }\n }\n}\n\n/**\n * Add an event listener to a selector\n * and returns a remove listener function.\n *\n * @param {String} selector\n * @param {String} type\n * @param {Function} callback\n * @return {Object}\n */\nfunction listenSelector(selector, type, callback) {\n return delegate(document.body, selector, type, callback);\n}\n\nmodule.exports = listen;\n\n\n/***/ }),\n\n/***/ 817:\n/***/ (function(module) {\n\nfunction select(element) {\n var selectedText;\n\n if (element.nodeName === 'SELECT') {\n element.focus();\n\n selectedText = element.value;\n }\n else if (element.nodeName === 'INPUT' || element.nodeName === 'TEXTAREA') {\n var isReadOnly = element.hasAttribute('readonly');\n\n if (!isReadOnly) {\n element.setAttribute('readonly', '');\n }\n\n element.select();\n element.setSelectionRange(0, element.value.length);\n\n if (!isReadOnly) {\n element.removeAttribute('readonly');\n }\n\n selectedText = element.value;\n }\n else {\n if (element.hasAttribute('contenteditable')) {\n element.focus();\n }\n\n var selection = window.getSelection();\n var range = document.createRange();\n\n range.selectNodeContents(element);\n selection.removeAllRanges();\n selection.addRange(range);\n\n selectedText = selection.toString();\n }\n\n return selectedText;\n}\n\nmodule.exports = select;\n\n\n/***/ }),\n\n/***/ 279:\n/***/ (function(module) {\n\nfunction E () {\n // Keep this empty so it's easier to inherit from\n // (via https://github.com/lipsmack from https://github.com/scottcorgan/tiny-emitter/issues/3)\n}\n\nE.prototype = {\n on: function (name, callback, ctx) {\n var e = this.e || (this.e = {});\n\n (e[name] || (e[name] = [])).push({\n fn: callback,\n ctx: ctx\n });\n\n return this;\n },\n\n once: function (name, callback, ctx) {\n var self = this;\n function listener () {\n self.off(name, listener);\n callback.apply(ctx, arguments);\n };\n\n listener._ = callback\n return this.on(name, listener, ctx);\n },\n\n emit: function (name) {\n var data = [].slice.call(arguments, 1);\n var evtArr = ((this.e || (this.e = {}))[name] || []).slice();\n var i = 0;\n var len = evtArr.length;\n\n for (i; i < len; i++) {\n evtArr[i].fn.apply(evtArr[i].ctx, data);\n }\n\n return this;\n },\n\n off: function (name, callback) {\n var e = this.e || (this.e = {});\n var evts = e[name];\n var liveEvents = [];\n\n if (evts && callback) {\n for (var i = 0, len = evts.length; i < len; i++) {\n if (evts[i].fn !== callback && evts[i].fn._ !== callback)\n liveEvents.push(evts[i]);\n }\n }\n\n // Remove event from queue to prevent memory leak\n // Suggested by https://github.com/lazd\n // Ref: https://github.com/scottcorgan/tiny-emitter/commit/c6ebfaa9bc973b33d110a84a307742b7cf94c953#commitcomment-5024910\n\n (liveEvents.length)\n ? e[name] = liveEvents\n : delete e[name];\n\n return this;\n }\n};\n\nmodule.exports = E;\nmodule.exports.TinyEmitter = E;\n\n\n/***/ })\n\n/******/ \t});\n/************************************************************************/\n/******/ \t// The module cache\n/******/ \tvar __webpack_module_cache__ = {};\n/******/ \t\n/******/ \t// The require function\n/******/ \tfunction __webpack_require__(moduleId) {\n/******/ \t\t// Check if module is in cache\n/******/ \t\tif(__webpack_module_cache__[moduleId]) {\n/******/ \t\t\treturn __webpack_module_cache__[moduleId].exports;\n/******/ \t\t}\n/******/ \t\t// Create a new module (and put it into the cache)\n/******/ \t\tvar module = __webpack_module_cache__[moduleId] = {\n/******/ \t\t\t// no module.id needed\n/******/ \t\t\t// no module.loaded needed\n/******/ \t\t\texports: {}\n/******/ \t\t};\n/******/ \t\n/******/ \t\t// Execute the module function\n/******/ \t\t__webpack_modules__[moduleId](module, module.exports, __webpack_require__);\n/******/ \t\n/******/ \t\t// Return the exports of the module\n/******/ \t\treturn module.exports;\n/******/ \t}\n/******/ \t\n/************************************************************************/\n/******/ \t/* webpack/runtime/compat get default export */\n/******/ \t!function() {\n/******/ \t\t// getDefaultExport function for compatibility with non-harmony modules\n/******/ \t\t__webpack_require__.n = function(module) {\n/******/ \t\t\tvar getter = module && module.__esModule ?\n/******/ \t\t\t\tfunction() { return module['default']; } :\n/******/ \t\t\t\tfunction() { return module; };\n/******/ \t\t\t__webpack_require__.d(getter, { a: getter });\n/******/ \t\t\treturn getter;\n/******/ \t\t};\n/******/ \t}();\n/******/ \t\n/******/ \t/* webpack/runtime/define property getters */\n/******/ \t!function() {\n/******/ \t\t// define getter functions for harmony exports\n/******/ \t\t__webpack_require__.d = function(exports, definition) {\n/******/ \t\t\tfor(var key in definition) {\n/******/ \t\t\t\tif(__webpack_require__.o(definition, key) && !__webpack_require__.o(exports, key)) {\n/******/ \t\t\t\t\tObject.defineProperty(exports, key, { enumerable: true, get: definition[key] });\n/******/ \t\t\t\t}\n/******/ \t\t\t}\n/******/ \t\t};\n/******/ \t}();\n/******/ \t\n/******/ \t/* webpack/runtime/hasOwnProperty shorthand */\n/******/ \t!function() {\n/******/ \t\t__webpack_require__.o = function(obj, prop) { return Object.prototype.hasOwnProperty.call(obj, prop); }\n/******/ \t}();\n/******/ \t\n/************************************************************************/\n/******/ \t// module exports must be returned from runtime so entry inlining is disabled\n/******/ \t// startup\n/******/ \t// Load entry module and return exports\n/******/ \treturn __webpack_require__(686);\n/******/ })()\n.default;\n});", "/*\n * Copyright (c) 2016-2025 Martin Donath \n *\n * Permission is hereby granted, free of charge, to any person obtaining a copy\n * of this software and associated documentation files (the \"Software\"), to\n * deal in the Software without restriction, including without limitation the\n * rights to use, copy, modify, merge, publish, distribute, sublicense, and/or\n * sell copies of the Software, and to permit persons to whom the Software is\n * furnished to do so, subject to the following conditions:\n *\n * The above copyright notice and this permission notice shall be included in\n * all copies or substantial portions of the Software.\n *\n * THE SOFTWARE IS PROVIDED \"AS IS\", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR\n * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,\n * FITNESS FOR A PARTICULAR PURPOSE AND NON-INFRINGEMENT. IN NO EVENT SHALL THE\n * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER\n * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING\n * FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS\n * IN THE SOFTWARE.\n */\n\nimport \"focus-visible\"\n\nimport {\n EMPTY,\n NEVER,\n Observable,\n Subject,\n defer,\n delay,\n filter,\n map,\n merge,\n mergeWith,\n shareReplay,\n switchMap\n} from \"rxjs\"\n\nimport { configuration, feature } from \"./_\"\nimport {\n at,\n getActiveElement,\n getOptionalElement,\n requestJSON,\n setLocation,\n setToggle,\n watchDocument,\n watchKeyboard,\n watchLocation,\n watchLocationTarget,\n watchMedia,\n watchPrint,\n watchScript,\n watchViewport\n} from \"./browser\"\nimport {\n getComponentElement,\n getComponentElements,\n mountAnnounce,\n mountBackToTop,\n mountConsent,\n mountContent,\n mountDialog,\n mountHeader,\n mountHeaderTitle,\n mountPalette,\n mountProgress,\n mountSearch,\n mountSearchHiglight,\n mountSidebar,\n mountSource,\n mountTableOfContents,\n mountTabs,\n watchHeader,\n watchMain\n} from \"./components\"\nimport {\n SearchIndex,\n setupClipboardJS,\n setupInstantNavigation,\n setupVersionSelector\n} from \"./integrations\"\nimport {\n patchEllipsis,\n patchIndeterminate,\n patchScrollfix,\n patchScrolllock\n} from \"./patches\"\nimport \"./polyfills\"\n\n/* ----------------------------------------------------------------------------\n * Functions - @todo refactor\n * ------------------------------------------------------------------------- */\n\n/**\n * Fetch search index\n *\n * @returns Search index observable\n */\nfunction fetchSearchIndex(): Observable {\n if (location.protocol === \"file:\") {\n return watchScript(\n `${new URL(\"search/search_index.js\", config.base)}`\n )\n .pipe(\n // @ts-ignore - @todo fix typings\n map(() => __index),\n shareReplay(1)\n )\n } else {\n return requestJSON(\n new URL(\"search/search_index.json\", config.base)\n )\n }\n}\n\n/* ----------------------------------------------------------------------------\n * Application\n * ------------------------------------------------------------------------- */\n\n/* Yay, JavaScript is available */\ndocument.documentElement.classList.remove(\"no-js\")\ndocument.documentElement.classList.add(\"js\")\n\n/* Set up navigation observables and subjects */\nconst document$ = watchDocument()\nconst location$ = watchLocation()\nconst target$ = watchLocationTarget(location$)\nconst keyboard$ = watchKeyboard()\n\n/* Set up media observables */\nconst viewport$ = watchViewport()\nconst tablet$ = watchMedia(\"(min-width: 60em)\")\nconst screen$ = watchMedia(\"(min-width: 76.25em)\")\nconst print$ = watchPrint()\n\n/* Retrieve search index, if search is enabled */\nconst config = configuration()\nconst index$ = document.forms.namedItem(\"search\")\n ? fetchSearchIndex()\n : NEVER\n\n/* Set up Clipboard.js integration */\nconst alert$ = new Subject()\nsetupClipboardJS({ alert$ })\n\n/* Set up progress indicator */\nconst progress$ = new Subject()\n\n/* Set up instant navigation, if enabled */\nif (feature(\"navigation.instant\"))\n setupInstantNavigation({ location$, viewport$, progress$ })\n .subscribe(document$)\n\n/* Set up version selector */\nif (config.version?.provider === \"mike\")\n setupVersionSelector({ document$ })\n\n/* Always close drawer and search on navigation */\nmerge(location$, target$)\n .pipe(\n delay(125)\n )\n .subscribe(() => {\n setToggle(\"drawer\", false)\n setToggle(\"search\", false)\n })\n\n/* Set up global keyboard handlers */\nkeyboard$\n .pipe(\n filter(({ mode }) => mode === \"global\")\n )\n .subscribe(key => {\n switch (key.type) {\n\n /* Go to previous page */\n case \"p\":\n case \",\":\n const prev = getOptionalElement(\"link[rel=prev]\")\n if (typeof prev !== \"undefined\")\n setLocation(prev)\n break\n\n /* Go to next page */\n case \"n\":\n case \".\":\n const next = getOptionalElement(\"link[rel=next]\")\n if (typeof next !== \"undefined\")\n setLocation(next)\n break\n\n /* Expand navigation, see https://bit.ly/3ZjG5io */\n case \"Enter\":\n const active = getActiveElement()\n if (active instanceof HTMLLabelElement)\n active.click()\n }\n })\n\n/* Set up patches */\npatchEllipsis({ viewport$, document$ })\npatchIndeterminate({ document$, tablet$ })\npatchScrollfix({ document$ })\npatchScrolllock({ viewport$, tablet$ })\n\n/* Set up header and main area observable */\nconst header$ = watchHeader(getComponentElement(\"header\"), { viewport$ })\nconst main$ = document$\n .pipe(\n map(() => getComponentElement(\"main\")),\n switchMap(el => watchMain(el, { viewport$, header$ })),\n shareReplay(1)\n )\n\n/* Set up control component observables */\nconst control$ = merge(\n\n /* Consent */\n ...getComponentElements(\"consent\")\n .map(el => mountConsent(el, { target$ })),\n\n /* Dialog */\n ...getComponentElements(\"dialog\")\n .map(el => mountDialog(el, { alert$ })),\n\n /* Color palette */\n ...getComponentElements(\"palette\")\n .map(el => mountPalette(el)),\n\n /* Progress bar */\n ...getComponentElements(\"progress\")\n .map(el => mountProgress(el, { progress$ })),\n\n /* Search */\n ...getComponentElements(\"search\")\n .map(el => mountSearch(el, { index$, keyboard$ })),\n\n /* Repository information */\n ...getComponentElements(\"source\")\n .map(el => mountSource(el))\n)\n\n/* Set up content component observables */\nconst content$ = defer(() => merge(\n\n /* Announcement bar */\n ...getComponentElements(\"announce\")\n .map(el => mountAnnounce(el)),\n\n /* Content */\n ...getComponentElements(\"content\")\n .map(el => mountContent(el, { viewport$, target$, print$ })),\n\n /* Search highlighting */\n ...getComponentElements(\"content\")\n .map(el => feature(\"search.highlight\")\n ? mountSearchHiglight(el, { index$, location$ })\n : EMPTY\n ),\n\n /* Header */\n ...getComponentElements(\"header\")\n .map(el => mountHeader(el, { viewport$, header$, main$ })),\n\n /* Header title */\n ...getComponentElements(\"header-title\")\n .map(el => mountHeaderTitle(el, { viewport$, header$ })),\n\n /* Sidebar */\n ...getComponentElements(\"sidebar\")\n .map(el => el.getAttribute(\"data-md-type\") === \"navigation\"\n ? at(screen$, () => mountSidebar(el, { viewport$, header$, main$ }))\n : at(tablet$, () => mountSidebar(el, { viewport$, header$, main$ }))\n ),\n\n /* Navigation tabs */\n ...getComponentElements(\"tabs\")\n .map(el => mountTabs(el, { viewport$, header$ })),\n\n /* Table of contents */\n ...getComponentElements(\"toc\")\n .map(el => mountTableOfContents(el, {\n viewport$, header$, main$, target$\n })),\n\n /* Back-to-top button */\n ...getComponentElements(\"top\")\n .map(el => mountBackToTop(el, { viewport$, header$, main$, target$ }))\n))\n\n/* Set up component observables */\nconst component$ = document$\n .pipe(\n switchMap(() => content$),\n mergeWith(control$),\n shareReplay(1)\n )\n\n/* Subscribe to all components */\ncomponent$.subscribe()\n\n/* ----------------------------------------------------------------------------\n * Exports\n * ------------------------------------------------------------------------- */\n\nwindow.document$ = document$ /* Document observable */\nwindow.location$ = location$ /* Location subject */\nwindow.target$ = target$ /* Location target observable */\nwindow.keyboard$ = keyboard$ /* Keyboard observable */\nwindow.viewport$ = viewport$ /* Viewport observable */\nwindow.tablet$ = tablet$ /* Media tablet observable */\nwindow.screen$ = screen$ /* Media screen observable */\nwindow.print$ = print$ /* Media print observable */\nwindow.alert$ = alert$ /* Alert subject */\nwindow.progress$ = progress$ /* Progress indicator subject */\nwindow.component$ = component$ /* Component observable */\n", "/******************************************************************************\nCopyright (c) Microsoft Corporation.\n\nPermission to use, copy, modify, and/or distribute this software for any\npurpose with or without fee is hereby granted.\n\nTHE SOFTWARE IS PROVIDED \"AS IS\" AND THE AUTHOR DISCLAIMS ALL WARRANTIES WITH\nREGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES OF MERCHANTABILITY\nAND FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR ANY SPECIAL, DIRECT,\nINDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES WHATSOEVER RESULTING FROM\nLOSS OF USE, DATA OR PROFITS, WHETHER IN AN ACTION OF CONTRACT, NEGLIGENCE OR\nOTHER TORTIOUS ACTION, ARISING OUT OF OR IN CONNECTION WITH THE USE OR\nPERFORMANCE OF THIS SOFTWARE.\n***************************************************************************** */\n/* global Reflect, Promise, SuppressedError, Symbol, Iterator */\n\nvar extendStatics = function(d, b) {\n extendStatics = Object.setPrototypeOf ||\n ({ __proto__: [] } instanceof Array && function (d, b) { d.__proto__ = b; }) ||\n function (d, b) { for (var p in b) if (Object.prototype.hasOwnProperty.call(b, p)) d[p] = b[p]; };\n return extendStatics(d, b);\n};\n\nexport function __extends(d, b) {\n if (typeof b !== \"function\" && b !== null)\n throw new TypeError(\"Class extends value \" + String(b) + \" is not a constructor or null\");\n extendStatics(d, b);\n function __() { this.constructor = d; }\n d.prototype = b === null ? Object.create(b) : (__.prototype = b.prototype, new __());\n}\n\nexport var __assign = function() {\n __assign = Object.assign || function __assign(t) {\n for (var s, i = 1, n = arguments.length; i < n; i++) {\n s = arguments[i];\n for (var p in s) if (Object.prototype.hasOwnProperty.call(s, p)) t[p] = s[p];\n }\n return t;\n }\n return __assign.apply(this, arguments);\n}\n\nexport function __rest(s, e) {\n var t = {};\n for (var p in s) if (Object.prototype.hasOwnProperty.call(s, p) && e.indexOf(p) < 0)\n t[p] = s[p];\n if (s != null && typeof Object.getOwnPropertySymbols === \"function\")\n for (var i = 0, p = Object.getOwnPropertySymbols(s); i < p.length; i++) {\n if (e.indexOf(p[i]) < 0 && Object.prototype.propertyIsEnumerable.call(s, p[i]))\n t[p[i]] = s[p[i]];\n }\n return t;\n}\n\nexport function __decorate(decorators, target, key, desc) {\n var c = arguments.length, r = c < 3 ? target : desc === null ? desc = Object.getOwnPropertyDescriptor(target, key) : desc, d;\n if (typeof Reflect === \"object\" && typeof Reflect.decorate === \"function\") r = Reflect.decorate(decorators, target, key, desc);\n else for (var i = decorators.length - 1; i >= 0; i--) if (d = decorators[i]) r = (c < 3 ? d(r) : c > 3 ? d(target, key, r) : d(target, key)) || r;\n return c > 3 && r && Object.defineProperty(target, key, r), r;\n}\n\nexport function __param(paramIndex, decorator) {\n return function (target, key) { decorator(target, key, paramIndex); }\n}\n\nexport function __esDecorate(ctor, descriptorIn, decorators, contextIn, initializers, extraInitializers) {\n function accept(f) { if (f !== void 0 && typeof f !== \"function\") throw new TypeError(\"Function expected\"); return f; }\n var kind = contextIn.kind, key = kind === \"getter\" ? \"get\" : kind === \"setter\" ? \"set\" : \"value\";\n var target = !descriptorIn && ctor ? contextIn[\"static\"] ? ctor : ctor.prototype : null;\n var descriptor = descriptorIn || (target ? Object.getOwnPropertyDescriptor(target, contextIn.name) : {});\n var _, done = false;\n for (var i = decorators.length - 1; i >= 0; i--) {\n var context = {};\n for (var p in contextIn) context[p] = p === \"access\" ? {} : contextIn[p];\n for (var p in contextIn.access) context.access[p] = contextIn.access[p];\n context.addInitializer = function (f) { if (done) throw new TypeError(\"Cannot add initializers after decoration has completed\"); extraInitializers.push(accept(f || null)); };\n var result = (0, decorators[i])(kind === \"accessor\" ? { get: descriptor.get, set: descriptor.set } : descriptor[key], context);\n if (kind === \"accessor\") {\n if (result === void 0) continue;\n if (result === null || typeof result !== \"object\") throw new TypeError(\"Object expected\");\n if (_ = accept(result.get)) descriptor.get = _;\n if (_ = accept(result.set)) descriptor.set = _;\n if (_ = accept(result.init)) initializers.unshift(_);\n }\n else if (_ = accept(result)) {\n if (kind === \"field\") initializers.unshift(_);\n else descriptor[key] = _;\n }\n }\n if (target) Object.defineProperty(target, contextIn.name, descriptor);\n done = true;\n};\n\nexport function __runInitializers(thisArg, initializers, value) {\n var useValue = arguments.length > 2;\n for (var i = 0; i < initializers.length; i++) {\n value = useValue ? initializers[i].call(thisArg, value) : initializers[i].call(thisArg);\n }\n return useValue ? value : void 0;\n};\n\nexport function __propKey(x) {\n return typeof x === \"symbol\" ? x : \"\".concat(x);\n};\n\nexport function __setFunctionName(f, name, prefix) {\n if (typeof name === \"symbol\") name = name.description ? \"[\".concat(name.description, \"]\") : \"\";\n return Object.defineProperty(f, \"name\", { configurable: true, value: prefix ? \"\".concat(prefix, \" \", name) : name });\n};\n\nexport function __metadata(metadataKey, metadataValue) {\n if (typeof Reflect === \"object\" && typeof Reflect.metadata === \"function\") return Reflect.metadata(metadataKey, metadataValue);\n}\n\nexport function __awaiter(thisArg, _arguments, P, generator) {\n function adopt(value) { return value instanceof P ? value : new P(function (resolve) { resolve(value); }); }\n return new (P || (P = Promise))(function (resolve, reject) {\n function fulfilled(value) { try { step(generator.next(value)); } catch (e) { reject(e); } }\n function rejected(value) { try { step(generator[\"throw\"](value)); } catch (e) { reject(e); } }\n function step(result) { result.done ? resolve(result.value) : adopt(result.value).then(fulfilled, rejected); }\n step((generator = generator.apply(thisArg, _arguments || [])).next());\n });\n}\n\nexport function __generator(thisArg, body) {\n var _ = { label: 0, sent: function() { if (t[0] & 1) throw t[1]; return t[1]; }, trys: [], ops: [] }, f, y, t, g = Object.create((typeof Iterator === \"function\" ? Iterator : Object).prototype);\n return g.next = verb(0), g[\"throw\"] = verb(1), g[\"return\"] = verb(2), typeof Symbol === \"function\" && (g[Symbol.iterator] = function() { return this; }), g;\n function verb(n) { return function (v) { return step([n, v]); }; }\n function step(op) {\n if (f) throw new TypeError(\"Generator is already executing.\");\n while (g && (g = 0, op[0] && (_ = 0)), _) try {\n if (f = 1, y && (t = op[0] & 2 ? y[\"return\"] : op[0] ? y[\"throw\"] || ((t = y[\"return\"]) && t.call(y), 0) : y.next) && !(t = t.call(y, op[1])).done) return t;\n if (y = 0, t) op = [op[0] & 2, t.value];\n switch (op[0]) {\n case 0: case 1: t = op; break;\n case 4: _.label++; return { value: op[1], done: false };\n case 5: _.label++; y = op[1]; op = [0]; continue;\n case 7: op = _.ops.pop(); _.trys.pop(); continue;\n default:\n if (!(t = _.trys, t = t.length > 0 && t[t.length - 1]) && (op[0] === 6 || op[0] === 2)) { _ = 0; continue; }\n if (op[0] === 3 && (!t || (op[1] > t[0] && op[1] < t[3]))) { _.label = op[1]; break; }\n if (op[0] === 6 && _.label < t[1]) { _.label = t[1]; t = op; break; }\n if (t && _.label < t[2]) { _.label = t[2]; _.ops.push(op); break; }\n if (t[2]) _.ops.pop();\n _.trys.pop(); continue;\n }\n op = body.call(thisArg, _);\n } catch (e) { op = [6, e]; y = 0; } finally { f = t = 0; }\n if (op[0] & 5) throw op[1]; return { value: op[0] ? op[1] : void 0, done: true };\n }\n}\n\nexport var __createBinding = Object.create ? (function(o, m, k, k2) {\n if (k2 === undefined) k2 = k;\n var desc = Object.getOwnPropertyDescriptor(m, k);\n if (!desc || (\"get\" in desc ? !m.__esModule : desc.writable || desc.configurable)) {\n desc = { enumerable: true, get: function() { return m[k]; } };\n }\n Object.defineProperty(o, k2, desc);\n}) : (function(o, m, k, k2) {\n if (k2 === undefined) k2 = k;\n o[k2] = m[k];\n});\n\nexport function __exportStar(m, o) {\n for (var p in m) if (p !== \"default\" && !Object.prototype.hasOwnProperty.call(o, p)) __createBinding(o, m, p);\n}\n\nexport function __values(o) {\n var s = typeof Symbol === \"function\" && Symbol.iterator, m = s && o[s], i = 0;\n if (m) return m.call(o);\n if (o && typeof o.length === \"number\") return {\n next: function () {\n if (o && i >= o.length) o = void 0;\n return { value: o && o[i++], done: !o };\n }\n };\n throw new TypeError(s ? \"Object is not iterable.\" : \"Symbol.iterator is not defined.\");\n}\n\nexport function __read(o, n) {\n var m = typeof Symbol === \"function\" && o[Symbol.iterator];\n if (!m) return o;\n var i = m.call(o), r, ar = [], e;\n try {\n while ((n === void 0 || n-- > 0) && !(r = i.next()).done) ar.push(r.value);\n }\n catch (error) { e = { error: error }; }\n finally {\n try {\n if (r && !r.done && (m = i[\"return\"])) m.call(i);\n }\n finally { if (e) throw e.error; }\n }\n return ar;\n}\n\n/** @deprecated */\nexport function __spread() {\n for (var ar = [], i = 0; i < arguments.length; i++)\n ar = ar.concat(__read(arguments[i]));\n return ar;\n}\n\n/** @deprecated */\nexport function __spreadArrays() {\n for (var s = 0, i = 0, il = arguments.length; i < il; i++) s += arguments[i].length;\n for (var r = Array(s), k = 0, i = 0; i < il; i++)\n for (var a = arguments[i], j = 0, jl = a.length; j < jl; j++, k++)\n r[k] = a[j];\n return r;\n}\n\nexport function __spreadArray(to, from, pack) {\n if (pack || arguments.length === 2) for (var i = 0, l = from.length, ar; i < l; i++) {\n if (ar || !(i in from)) {\n if (!ar) ar = Array.prototype.slice.call(from, 0, i);\n ar[i] = from[i];\n }\n }\n return to.concat(ar || Array.prototype.slice.call(from));\n}\n\nexport function __await(v) {\n return this instanceof __await ? (this.v = v, this) : new __await(v);\n}\n\nexport function __asyncGenerator(thisArg, _arguments, generator) {\n if (!Symbol.asyncIterator) throw new TypeError(\"Symbol.asyncIterator is not defined.\");\n var g = generator.apply(thisArg, _arguments || []), i, q = [];\n return i = Object.create((typeof AsyncIterator === \"function\" ? AsyncIterator : Object).prototype), verb(\"next\"), verb(\"throw\"), verb(\"return\", awaitReturn), i[Symbol.asyncIterator] = function () { return this; }, i;\n function awaitReturn(f) { return function (v) { return Promise.resolve(v).then(f, reject); }; }\n function verb(n, f) { if (g[n]) { i[n] = function (v) { return new Promise(function (a, b) { q.push([n, v, a, b]) > 1 || resume(n, v); }); }; if (f) i[n] = f(i[n]); } }\n function resume(n, v) { try { step(g[n](v)); } catch (e) { settle(q[0][3], e); } }\n function step(r) { r.value instanceof __await ? Promise.resolve(r.value.v).then(fulfill, reject) : settle(q[0][2], r); }\n function fulfill(value) { resume(\"next\", value); }\n function reject(value) { resume(\"throw\", value); }\n function settle(f, v) { if (f(v), q.shift(), q.length) resume(q[0][0], q[0][1]); }\n}\n\nexport function __asyncDelegator(o) {\n var i, p;\n return i = {}, verb(\"next\"), verb(\"throw\", function (e) { throw e; }), verb(\"return\"), i[Symbol.iterator] = function () { return this; }, i;\n function verb(n, f) { i[n] = o[n] ? function (v) { return (p = !p) ? { value: __await(o[n](v)), done: false } : f ? f(v) : v; } : f; }\n}\n\nexport function __asyncValues(o) {\n if (!Symbol.asyncIterator) throw new TypeError(\"Symbol.asyncIterator is not defined.\");\n var m = o[Symbol.asyncIterator], i;\n return m ? m.call(o) : (o = typeof __values === \"function\" ? __values(o) : o[Symbol.iterator](), i = {}, verb(\"next\"), verb(\"throw\"), verb(\"return\"), i[Symbol.asyncIterator] = function () { return this; }, i);\n function verb(n) { i[n] = o[n] && function (v) { return new Promise(function (resolve, reject) { v = o[n](v), settle(resolve, reject, v.done, v.value); }); }; }\n function settle(resolve, reject, d, v) { Promise.resolve(v).then(function(v) { resolve({ value: v, done: d }); }, reject); }\n}\n\nexport function __makeTemplateObject(cooked, raw) {\n if (Object.defineProperty) { Object.defineProperty(cooked, \"raw\", { value: raw }); } else { cooked.raw = raw; }\n return cooked;\n};\n\nvar __setModuleDefault = Object.create ? (function(o, v) {\n Object.defineProperty(o, \"default\", { enumerable: true, value: v });\n}) : function(o, v) {\n o[\"default\"] = v;\n};\n\nexport function __importStar(mod) {\n if (mod && mod.__esModule) return mod;\n var result = {};\n if (mod != null) for (var k in mod) if (k !== \"default\" && Object.prototype.hasOwnProperty.call(mod, k)) __createBinding(result, mod, k);\n __setModuleDefault(result, mod);\n return result;\n}\n\nexport function __importDefault(mod) {\n return (mod && mod.__esModule) ? mod : { default: mod };\n}\n\nexport function __classPrivateFieldGet(receiver, state, kind, f) {\n if (kind === \"a\" && !f) throw new TypeError(\"Private accessor was defined without a getter\");\n if (typeof state === \"function\" ? receiver !== state || !f : !state.has(receiver)) throw new TypeError(\"Cannot read private member from an object whose class did not declare it\");\n return kind === \"m\" ? f : kind === \"a\" ? f.call(receiver) : f ? f.value : state.get(receiver);\n}\n\nexport function __classPrivateFieldSet(receiver, state, value, kind, f) {\n if (kind === \"m\") throw new TypeError(\"Private method is not writable\");\n if (kind === \"a\" && !f) throw new TypeError(\"Private accessor was defined without a setter\");\n if (typeof state === \"function\" ? receiver !== state || !f : !state.has(receiver)) throw new TypeError(\"Cannot write private member to an object whose class did not declare it\");\n return (kind === \"a\" ? f.call(receiver, value) : f ? f.value = value : state.set(receiver, value)), value;\n}\n\nexport function __classPrivateFieldIn(state, receiver) {\n if (receiver === null || (typeof receiver !== \"object\" && typeof receiver !== \"function\")) throw new TypeError(\"Cannot use 'in' operator on non-object\");\n return typeof state === \"function\" ? receiver === state : state.has(receiver);\n}\n\nexport function __addDisposableResource(env, value, async) {\n if (value !== null && value !== void 0) {\n if (typeof value !== \"object\" && typeof value !== \"function\") throw new TypeError(\"Object expected.\");\n var dispose, inner;\n if (async) {\n if (!Symbol.asyncDispose) throw new TypeError(\"Symbol.asyncDispose is not defined.\");\n dispose = value[Symbol.asyncDispose];\n }\n if (dispose === void 0) {\n if (!Symbol.dispose) throw new TypeError(\"Symbol.dispose is not defined.\");\n dispose = value[Symbol.dispose];\n if (async) inner = dispose;\n }\n if (typeof dispose !== \"function\") throw new TypeError(\"Object not disposable.\");\n if (inner) dispose = function() { try { inner.call(this); } catch (e) { return Promise.reject(e); } };\n env.stack.push({ value: value, dispose: dispose, async: async });\n }\n else if (async) {\n env.stack.push({ async: true });\n }\n return value;\n}\n\nvar _SuppressedError = typeof SuppressedError === \"function\" ? SuppressedError : function (error, suppressed, message) {\n var e = new Error(message);\n return e.name = \"SuppressedError\", e.error = error, e.suppressed = suppressed, e;\n};\n\nexport function __disposeResources(env) {\n function fail(e) {\n env.error = env.hasError ? new _SuppressedError(e, env.error, \"An error was suppressed during disposal.\") : e;\n env.hasError = true;\n }\n var r, s = 0;\n function next() {\n while (r = env.stack.pop()) {\n try {\n if (!r.async && s === 1) return s = 0, env.stack.push(r), Promise.resolve().then(next);\n if (r.dispose) {\n var result = r.dispose.call(r.value);\n if (r.async) return s |= 2, Promise.resolve(result).then(next, function(e) { fail(e); return next(); });\n }\n else s |= 1;\n }\n catch (e) {\n fail(e);\n }\n }\n if (s === 1) return env.hasError ? Promise.reject(env.error) : Promise.resolve();\n if (env.hasError) throw env.error;\n }\n return next();\n}\n\nexport default {\n __extends,\n __assign,\n __rest,\n __decorate,\n __param,\n __metadata,\n __awaiter,\n __generator,\n __createBinding,\n __exportStar,\n __values,\n __read,\n __spread,\n __spreadArrays,\n __spreadArray,\n __await,\n __asyncGenerator,\n __asyncDelegator,\n __asyncValues,\n __makeTemplateObject,\n __importStar,\n __importDefault,\n __classPrivateFieldGet,\n __classPrivateFieldSet,\n __classPrivateFieldIn,\n __addDisposableResource,\n __disposeResources,\n};\n", "/**\n * Returns true if the object is a function.\n * @param value The value to check\n */\nexport function isFunction(value: any): value is (...args: any[]) => any {\n return typeof value === 'function';\n}\n", "/**\n * Used to create Error subclasses until the community moves away from ES5.\n *\n * This is because compiling from TypeScript down to ES5 has issues with subclassing Errors\n * as well as other built-in types: https://github.com/Microsoft/TypeScript/issues/12123\n *\n * @param createImpl A factory function to create the actual constructor implementation. The returned\n * function should be a named function that calls `_super` internally.\n */\nexport function createErrorClass(createImpl: (_super: any) => any): T {\n const _super = (instance: any) => {\n Error.call(instance);\n instance.stack = new Error().stack;\n };\n\n const ctorFunc = createImpl(_super);\n ctorFunc.prototype = Object.create(Error.prototype);\n ctorFunc.prototype.constructor = ctorFunc;\n return ctorFunc;\n}\n", "import { createErrorClass } from './createErrorClass';\n\nexport interface UnsubscriptionError extends Error {\n readonly errors: any[];\n}\n\nexport interface UnsubscriptionErrorCtor {\n /**\n * @deprecated Internal implementation detail. Do not construct error instances.\n * Cannot be tagged as internal: https://github.com/ReactiveX/rxjs/issues/6269\n */\n new (errors: any[]): UnsubscriptionError;\n}\n\n/**\n * An error thrown when one or more errors have occurred during the\n * `unsubscribe` of a {@link Subscription}.\n */\nexport const UnsubscriptionError: UnsubscriptionErrorCtor = createErrorClass(\n (_super) =>\n function UnsubscriptionErrorImpl(this: any, errors: (Error | string)[]) {\n _super(this);\n this.message = errors\n ? `${errors.length} errors occurred during unsubscription:\n${errors.map((err, i) => `${i + 1}) ${err.toString()}`).join('\\n ')}`\n : '';\n this.name = 'UnsubscriptionError';\n this.errors = errors;\n }\n);\n", "/**\n * Removes an item from an array, mutating it.\n * @param arr The array to remove the item from\n * @param item The item to remove\n */\nexport function arrRemove(arr: T[] | undefined | null, item: T) {\n if (arr) {\n const index = arr.indexOf(item);\n 0 <= index && arr.splice(index, 1);\n }\n}\n", "import { isFunction } from './util/isFunction';\nimport { UnsubscriptionError } from './util/UnsubscriptionError';\nimport { SubscriptionLike, TeardownLogic, Unsubscribable } from './types';\nimport { arrRemove } from './util/arrRemove';\n\n/**\n * Represents a disposable resource, such as the execution of an Observable. A\n * Subscription has one important method, `unsubscribe`, that takes no argument\n * and just disposes the resource held by the subscription.\n *\n * Additionally, subscriptions may be grouped together through the `add()`\n * method, which will attach a child Subscription to the current Subscription.\n * When a Subscription is unsubscribed, all its children (and its grandchildren)\n * will be unsubscribed as well.\n */\nexport class Subscription implements SubscriptionLike {\n public static EMPTY = (() => {\n const empty = new Subscription();\n empty.closed = true;\n return empty;\n })();\n\n /**\n * A flag to indicate whether this Subscription has already been unsubscribed.\n */\n public closed = false;\n\n private _parentage: Subscription[] | Subscription | null = null;\n\n /**\n * The list of registered finalizers to execute upon unsubscription. Adding and removing from this\n * list occurs in the {@link #add} and {@link #remove} methods.\n */\n private _finalizers: Exclude[] | null = null;\n\n /**\n * @param initialTeardown A function executed first as part of the finalization\n * process that is kicked off when {@link #unsubscribe} is called.\n */\n constructor(private initialTeardown?: () => void) {}\n\n /**\n * Disposes the resources held by the subscription. May, for instance, cancel\n * an ongoing Observable execution or cancel any other type of work that\n * started when the Subscription was created.\n */\n unsubscribe(): void {\n let errors: any[] | undefined;\n\n if (!this.closed) {\n this.closed = true;\n\n // Remove this from it's parents.\n const { _parentage } = this;\n if (_parentage) {\n this._parentage = null;\n if (Array.isArray(_parentage)) {\n for (const parent of _parentage) {\n parent.remove(this);\n }\n } else {\n _parentage.remove(this);\n }\n }\n\n const { initialTeardown: initialFinalizer } = this;\n if (isFunction(initialFinalizer)) {\n try {\n initialFinalizer();\n } catch (e) {\n errors = e instanceof UnsubscriptionError ? e.errors : [e];\n }\n }\n\n const { _finalizers } = this;\n if (_finalizers) {\n this._finalizers = null;\n for (const finalizer of _finalizers) {\n try {\n execFinalizer(finalizer);\n } catch (err) {\n errors = errors ?? [];\n if (err instanceof UnsubscriptionError) {\n errors = [...errors, ...err.errors];\n } else {\n errors.push(err);\n }\n }\n }\n }\n\n if (errors) {\n throw new UnsubscriptionError(errors);\n }\n }\n }\n\n /**\n * Adds a finalizer to this subscription, so that finalization will be unsubscribed/called\n * when this subscription is unsubscribed. If this subscription is already {@link #closed},\n * because it has already been unsubscribed, then whatever finalizer is passed to it\n * will automatically be executed (unless the finalizer itself is also a closed subscription).\n *\n * Closed Subscriptions cannot be added as finalizers to any subscription. Adding a closed\n * subscription to a any subscription will result in no operation. (A noop).\n *\n * Adding a subscription to itself, or adding `null` or `undefined` will not perform any\n * operation at all. (A noop).\n *\n * `Subscription` instances that are added to this instance will automatically remove themselves\n * if they are unsubscribed. Functions and {@link Unsubscribable} objects that you wish to remove\n * will need to be removed manually with {@link #remove}\n *\n * @param teardown The finalization logic to add to this subscription.\n */\n add(teardown: TeardownLogic): void {\n // Only add the finalizer if it's not undefined\n // and don't add a subscription to itself.\n if (teardown && teardown !== this) {\n if (this.closed) {\n // If this subscription is already closed,\n // execute whatever finalizer is handed to it automatically.\n execFinalizer(teardown);\n } else {\n if (teardown instanceof Subscription) {\n // We don't add closed subscriptions, and we don't add the same subscription\n // twice. Subscription unsubscribe is idempotent.\n if (teardown.closed || teardown._hasParent(this)) {\n return;\n }\n teardown._addParent(this);\n }\n (this._finalizers = this._finalizers ?? []).push(teardown);\n }\n }\n }\n\n /**\n * Checks to see if a this subscription already has a particular parent.\n * This will signal that this subscription has already been added to the parent in question.\n * @param parent the parent to check for\n */\n private _hasParent(parent: Subscription) {\n const { _parentage } = this;\n return _parentage === parent || (Array.isArray(_parentage) && _parentage.includes(parent));\n }\n\n /**\n * Adds a parent to this subscription so it can be removed from the parent if it\n * unsubscribes on it's own.\n *\n * NOTE: THIS ASSUMES THAT {@link _hasParent} HAS ALREADY BEEN CHECKED.\n * @param parent The parent subscription to add\n */\n private _addParent(parent: Subscription) {\n const { _parentage } = this;\n this._parentage = Array.isArray(_parentage) ? (_parentage.push(parent), _parentage) : _parentage ? [_parentage, parent] : parent;\n }\n\n /**\n * Called on a child when it is removed via {@link #remove}.\n * @param parent The parent to remove\n */\n private _removeParent(parent: Subscription) {\n const { _parentage } = this;\n if (_parentage === parent) {\n this._parentage = null;\n } else if (Array.isArray(_parentage)) {\n arrRemove(_parentage, parent);\n }\n }\n\n /**\n * Removes a finalizer from this subscription that was previously added with the {@link #add} method.\n *\n * Note that `Subscription` instances, when unsubscribed, will automatically remove themselves\n * from every other `Subscription` they have been added to. This means that using the `remove` method\n * is not a common thing and should be used thoughtfully.\n *\n * If you add the same finalizer instance of a function or an unsubscribable object to a `Subscription` instance\n * more than once, you will need to call `remove` the same number of times to remove all instances.\n *\n * All finalizer instances are removed to free up memory upon unsubscription.\n *\n * @param teardown The finalizer to remove from this subscription\n */\n remove(teardown: Exclude): void {\n const { _finalizers } = this;\n _finalizers && arrRemove(_finalizers, teardown);\n\n if (teardown instanceof Subscription) {\n teardown._removeParent(this);\n }\n }\n}\n\nexport const EMPTY_SUBSCRIPTION = Subscription.EMPTY;\n\nexport function isSubscription(value: any): value is Subscription {\n return (\n value instanceof Subscription ||\n (value && 'closed' in value && isFunction(value.remove) && isFunction(value.add) && isFunction(value.unsubscribe))\n );\n}\n\nfunction execFinalizer(finalizer: Unsubscribable | (() => void)) {\n if (isFunction(finalizer)) {\n finalizer();\n } else {\n finalizer.unsubscribe();\n }\n}\n", "import { Subscriber } from './Subscriber';\nimport { ObservableNotification } from './types';\n\n/**\n * The {@link GlobalConfig} object for RxJS. It is used to configure things\n * like how to react on unhandled errors.\n */\nexport const config: GlobalConfig = {\n onUnhandledError: null,\n onStoppedNotification: null,\n Promise: undefined,\n useDeprecatedSynchronousErrorHandling: false,\n useDeprecatedNextContext: false,\n};\n\n/**\n * The global configuration object for RxJS, used to configure things\n * like how to react on unhandled errors. Accessible via {@link config}\n * object.\n */\nexport interface GlobalConfig {\n /**\n * A registration point for unhandled errors from RxJS. These are errors that\n * cannot were not handled by consuming code in the usual subscription path. For\n * example, if you have this configured, and you subscribe to an observable without\n * providing an error handler, errors from that subscription will end up here. This\n * will _always_ be called asynchronously on another job in the runtime. This is because\n * we do not want errors thrown in this user-configured handler to interfere with the\n * behavior of the library.\n */\n onUnhandledError: ((err: any) => void) | null;\n\n /**\n * A registration point for notifications that cannot be sent to subscribers because they\n * have completed, errored or have been explicitly unsubscribed. By default, next, complete\n * and error notifications sent to stopped subscribers are noops. However, sometimes callers\n * might want a different behavior. For example, with sources that attempt to report errors\n * to stopped subscribers, a caller can configure RxJS to throw an unhandled error instead.\n * This will _always_ be called asynchronously on another job in the runtime. This is because\n * we do not want errors thrown in this user-configured handler to interfere with the\n * behavior of the library.\n */\n onStoppedNotification: ((notification: ObservableNotification, subscriber: Subscriber) => void) | null;\n\n /**\n * The promise constructor used by default for {@link Observable#toPromise toPromise} and {@link Observable#forEach forEach}\n * methods.\n *\n * @deprecated As of version 8, RxJS will no longer support this sort of injection of a\n * Promise constructor. If you need a Promise implementation other than native promises,\n * please polyfill/patch Promise as you see appropriate. Will be removed in v8.\n */\n Promise?: PromiseConstructorLike;\n\n /**\n * If true, turns on synchronous error rethrowing, which is a deprecated behavior\n * in v6 and higher. This behavior enables bad patterns like wrapping a subscribe\n * call in a try/catch block. It also enables producer interference, a nasty bug\n * where a multicast can be broken for all observers by a downstream consumer with\n * an unhandled error. DO NOT USE THIS FLAG UNLESS IT'S NEEDED TO BUY TIME\n * FOR MIGRATION REASONS.\n *\n * @deprecated As of version 8, RxJS will no longer support synchronous throwing\n * of unhandled errors. All errors will be thrown on a separate call stack to prevent bad\n * behaviors described above. Will be removed in v8.\n */\n useDeprecatedSynchronousErrorHandling: boolean;\n\n /**\n * If true, enables an as-of-yet undocumented feature from v5: The ability to access\n * `unsubscribe()` via `this` context in `next` functions created in observers passed\n * to `subscribe`.\n *\n * This is being removed because the performance was severely problematic, and it could also cause\n * issues when types other than POJOs are passed to subscribe as subscribers, as they will likely have\n * their `this` context overwritten.\n *\n * @deprecated As of version 8, RxJS will no longer support altering the\n * context of next functions provided as part of an observer to Subscribe. Instead,\n * you will have access to a subscription or a signal or token that will allow you to do things like\n * unsubscribe and test closed status. Will be removed in v8.\n */\n useDeprecatedNextContext: boolean;\n}\n", "import type { TimerHandle } from './timerHandle';\ntype SetTimeoutFunction = (handler: () => void, timeout?: number, ...args: any[]) => TimerHandle;\ntype ClearTimeoutFunction = (handle: TimerHandle) => void;\n\ninterface TimeoutProvider {\n setTimeout: SetTimeoutFunction;\n clearTimeout: ClearTimeoutFunction;\n delegate:\n | {\n setTimeout: SetTimeoutFunction;\n clearTimeout: ClearTimeoutFunction;\n }\n | undefined;\n}\n\nexport const timeoutProvider: TimeoutProvider = {\n // When accessing the delegate, use the variable rather than `this` so that\n // the functions can be called without being bound to the provider.\n setTimeout(handler: () => void, timeout?: number, ...args) {\n const { delegate } = timeoutProvider;\n if (delegate?.setTimeout) {\n return delegate.setTimeout(handler, timeout, ...args);\n }\n return setTimeout(handler, timeout, ...args);\n },\n clearTimeout(handle) {\n const { delegate } = timeoutProvider;\n return (delegate?.clearTimeout || clearTimeout)(handle as any);\n },\n delegate: undefined,\n};\n", "import { config } from '../config';\nimport { timeoutProvider } from '../scheduler/timeoutProvider';\n\n/**\n * Handles an error on another job either with the user-configured {@link onUnhandledError},\n * or by throwing it on that new job so it can be picked up by `window.onerror`, `process.on('error')`, etc.\n *\n * This should be called whenever there is an error that is out-of-band with the subscription\n * or when an error hits a terminal boundary of the subscription and no error handler was provided.\n *\n * @param err the error to report\n */\nexport function reportUnhandledError(err: any) {\n timeoutProvider.setTimeout(() => {\n const { onUnhandledError } = config;\n if (onUnhandledError) {\n // Execute the user-configured error handler.\n onUnhandledError(err);\n } else {\n // Throw so it is picked up by the runtime's uncaught error mechanism.\n throw err;\n }\n });\n}\n", "/* tslint:disable:no-empty */\nexport function noop() { }\n", "import { CompleteNotification, NextNotification, ErrorNotification } from './types';\n\n/**\n * A completion object optimized for memory use and created to be the\n * same \"shape\" as other notifications in v8.\n * @internal\n */\nexport const COMPLETE_NOTIFICATION = (() => createNotification('C', undefined, undefined) as CompleteNotification)();\n\n/**\n * Internal use only. Creates an optimized error notification that is the same \"shape\"\n * as other notifications.\n * @internal\n */\nexport function errorNotification(error: any): ErrorNotification {\n return createNotification('E', undefined, error) as any;\n}\n\n/**\n * Internal use only. Creates an optimized next notification that is the same \"shape\"\n * as other notifications.\n * @internal\n */\nexport function nextNotification(value: T) {\n return createNotification('N', value, undefined) as NextNotification;\n}\n\n/**\n * Ensures that all notifications created internally have the same \"shape\" in v8.\n *\n * TODO: This is only exported to support a crazy legacy test in `groupBy`.\n * @internal\n */\nexport function createNotification(kind: 'N' | 'E' | 'C', value: any, error: any) {\n return {\n kind,\n value,\n error,\n };\n}\n", "import { config } from '../config';\n\nlet context: { errorThrown: boolean; error: any } | null = null;\n\n/**\n * Handles dealing with errors for super-gross mode. Creates a context, in which\n * any synchronously thrown errors will be passed to {@link captureError}. Which\n * will record the error such that it will be rethrown after the call back is complete.\n * TODO: Remove in v8\n * @param cb An immediately executed function.\n */\nexport function errorContext(cb: () => void) {\n if (config.useDeprecatedSynchronousErrorHandling) {\n const isRoot = !context;\n if (isRoot) {\n context = { errorThrown: false, error: null };\n }\n cb();\n if (isRoot) {\n const { errorThrown, error } = context!;\n context = null;\n if (errorThrown) {\n throw error;\n }\n }\n } else {\n // This is the general non-deprecated path for everyone that\n // isn't crazy enough to use super-gross mode (useDeprecatedSynchronousErrorHandling)\n cb();\n }\n}\n\n/**\n * Captures errors only in super-gross mode.\n * @param err the error to capture\n */\nexport function captureError(err: any) {\n if (config.useDeprecatedSynchronousErrorHandling && context) {\n context.errorThrown = true;\n context.error = err;\n }\n}\n", "import { isFunction } from './util/isFunction';\nimport { Observer, ObservableNotification } from './types';\nimport { isSubscription, Subscription } from './Subscription';\nimport { config } from './config';\nimport { reportUnhandledError } from './util/reportUnhandledError';\nimport { noop } from './util/noop';\nimport { nextNotification, errorNotification, COMPLETE_NOTIFICATION } from './NotificationFactories';\nimport { timeoutProvider } from './scheduler/timeoutProvider';\nimport { captureError } from './util/errorContext';\n\n/**\n * Implements the {@link Observer} interface and extends the\n * {@link Subscription} class. While the {@link Observer} is the public API for\n * consuming the values of an {@link Observable}, all Observers get converted to\n * a Subscriber, in order to provide Subscription-like capabilities such as\n * `unsubscribe`. Subscriber is a common type in RxJS, and crucial for\n * implementing operators, but it is rarely used as a public API.\n */\nexport class Subscriber extends Subscription implements Observer {\n /**\n * A static factory for a Subscriber, given a (potentially partial) definition\n * of an Observer.\n * @param next The `next` callback of an Observer.\n * @param error The `error` callback of an\n * Observer.\n * @param complete The `complete` callback of an\n * Observer.\n * @return A Subscriber wrapping the (partially defined)\n * Observer represented by the given arguments.\n * @deprecated Do not use. Will be removed in v8. There is no replacement for this\n * method, and there is no reason to be creating instances of `Subscriber` directly.\n * If you have a specific use case, please file an issue.\n */\n static create(next?: (x?: T) => void, error?: (e?: any) => void, complete?: () => void): Subscriber {\n return new SafeSubscriber(next, error, complete);\n }\n\n /** @deprecated Internal implementation detail, do not use directly. Will be made internal in v8. */\n protected isStopped: boolean = false;\n /** @deprecated Internal implementation detail, do not use directly. Will be made internal in v8. */\n protected destination: Subscriber | Observer; // this `any` is the escape hatch to erase extra type param (e.g. R)\n\n /**\n * @deprecated Internal implementation detail, do not use directly. Will be made internal in v8.\n * There is no reason to directly create an instance of Subscriber. This type is exported for typings reasons.\n */\n constructor(destination?: Subscriber | Observer) {\n super();\n if (destination) {\n this.destination = destination;\n // Automatically chain subscriptions together here.\n // if destination is a Subscription, then it is a Subscriber.\n if (isSubscription(destination)) {\n destination.add(this);\n }\n } else {\n this.destination = EMPTY_OBSERVER;\n }\n }\n\n /**\n * The {@link Observer} callback to receive notifications of type `next` from\n * the Observable, with a value. The Observable may call this method 0 or more\n * times.\n * @param value The `next` value.\n */\n next(value: T): void {\n if (this.isStopped) {\n handleStoppedNotification(nextNotification(value), this);\n } else {\n this._next(value!);\n }\n }\n\n /**\n * The {@link Observer} callback to receive notifications of type `error` from\n * the Observable, with an attached `Error`. Notifies the Observer that\n * the Observable has experienced an error condition.\n * @param err The `error` exception.\n */\n error(err?: any): void {\n if (this.isStopped) {\n handleStoppedNotification(errorNotification(err), this);\n } else {\n this.isStopped = true;\n this._error(err);\n }\n }\n\n /**\n * The {@link Observer} callback to receive a valueless notification of type\n * `complete` from the Observable. Notifies the Observer that the Observable\n * has finished sending push-based notifications.\n */\n complete(): void {\n if (this.isStopped) {\n handleStoppedNotification(COMPLETE_NOTIFICATION, this);\n } else {\n this.isStopped = true;\n this._complete();\n }\n }\n\n unsubscribe(): void {\n if (!this.closed) {\n this.isStopped = true;\n super.unsubscribe();\n this.destination = null!;\n }\n }\n\n protected _next(value: T): void {\n this.destination.next(value);\n }\n\n protected _error(err: any): void {\n try {\n this.destination.error(err);\n } finally {\n this.unsubscribe();\n }\n }\n\n protected _complete(): void {\n try {\n this.destination.complete();\n } finally {\n this.unsubscribe();\n }\n }\n}\n\n/**\n * This bind is captured here because we want to be able to have\n * compatibility with monoid libraries that tend to use a method named\n * `bind`. In particular, a library called Monio requires this.\n */\nconst _bind = Function.prototype.bind;\n\nfunction bind any>(fn: Fn, thisArg: any): Fn {\n return _bind.call(fn, thisArg);\n}\n\n/**\n * Internal optimization only, DO NOT EXPOSE.\n * @internal\n */\nclass ConsumerObserver implements Observer {\n constructor(private partialObserver: Partial>) {}\n\n next(value: T): void {\n const { partialObserver } = this;\n if (partialObserver.next) {\n try {\n partialObserver.next(value);\n } catch (error) {\n handleUnhandledError(error);\n }\n }\n }\n\n error(err: any): void {\n const { partialObserver } = this;\n if (partialObserver.error) {\n try {\n partialObserver.error(err);\n } catch (error) {\n handleUnhandledError(error);\n }\n } else {\n handleUnhandledError(err);\n }\n }\n\n complete(): void {\n const { partialObserver } = this;\n if (partialObserver.complete) {\n try {\n partialObserver.complete();\n } catch (error) {\n handleUnhandledError(error);\n }\n }\n }\n}\n\nexport class SafeSubscriber extends Subscriber {\n constructor(\n observerOrNext?: Partial> | ((value: T) => void) | null,\n error?: ((e?: any) => void) | null,\n complete?: (() => void) | null\n ) {\n super();\n\n let partialObserver: Partial>;\n if (isFunction(observerOrNext) || !observerOrNext) {\n // The first argument is a function, not an observer. The next\n // two arguments *could* be observers, or they could be empty.\n partialObserver = {\n next: (observerOrNext ?? undefined) as ((value: T) => void) | undefined,\n error: error ?? undefined,\n complete: complete ?? undefined,\n };\n } else {\n // The first argument is a partial observer.\n let context: any;\n if (this && config.useDeprecatedNextContext) {\n // This is a deprecated path that made `this.unsubscribe()` available in\n // next handler functions passed to subscribe. This only exists behind a flag\n // now, as it is *very* slow.\n context = Object.create(observerOrNext);\n context.unsubscribe = () => this.unsubscribe();\n partialObserver = {\n next: observerOrNext.next && bind(observerOrNext.next, context),\n error: observerOrNext.error && bind(observerOrNext.error, context),\n complete: observerOrNext.complete && bind(observerOrNext.complete, context),\n };\n } else {\n // The \"normal\" path. Just use the partial observer directly.\n partialObserver = observerOrNext;\n }\n }\n\n // Wrap the partial observer to ensure it's a full observer, and\n // make sure proper error handling is accounted for.\n this.destination = new ConsumerObserver(partialObserver);\n }\n}\n\nfunction handleUnhandledError(error: any) {\n if (config.useDeprecatedSynchronousErrorHandling) {\n captureError(error);\n } else {\n // Ideal path, we report this as an unhandled error,\n // which is thrown on a new call stack.\n reportUnhandledError(error);\n }\n}\n\n/**\n * An error handler used when no error handler was supplied\n * to the SafeSubscriber -- meaning no error handler was supplied\n * do the `subscribe` call on our observable.\n * @param err The error to handle\n */\nfunction defaultErrorHandler(err: any) {\n throw err;\n}\n\n/**\n * A handler for notifications that cannot be sent to a stopped subscriber.\n * @param notification The notification being sent.\n * @param subscriber The stopped subscriber.\n */\nfunction handleStoppedNotification(notification: ObservableNotification, subscriber: Subscriber) {\n const { onStoppedNotification } = config;\n onStoppedNotification && timeoutProvider.setTimeout(() => onStoppedNotification(notification, subscriber));\n}\n\n/**\n * The observer used as a stub for subscriptions where the user did not\n * pass any arguments to `subscribe`. Comes with the default error handling\n * behavior.\n */\nexport const EMPTY_OBSERVER: Readonly> & { closed: true } = {\n closed: true,\n next: noop,\n error: defaultErrorHandler,\n complete: noop,\n};\n", "/**\n * Symbol.observable or a string \"@@observable\". Used for interop\n *\n * @deprecated We will no longer be exporting this symbol in upcoming versions of RxJS.\n * Instead polyfill and use Symbol.observable directly *or* use https://www.npmjs.com/package/symbol-observable\n */\nexport const observable: string | symbol = (() => (typeof Symbol === 'function' && Symbol.observable) || '@@observable')();\n", "/**\n * This function takes one parameter and just returns it. Simply put,\n * this is like `(x: T): T => x`.\n *\n * ## Examples\n *\n * This is useful in some cases when using things like `mergeMap`\n *\n * ```ts\n * import { interval, take, map, range, mergeMap, identity } from 'rxjs';\n *\n * const source$ = interval(1000).pipe(take(5));\n *\n * const result$ = source$.pipe(\n * map(i => range(i)),\n * mergeMap(identity) // same as mergeMap(x => x)\n * );\n *\n * result$.subscribe({\n * next: console.log\n * });\n * ```\n *\n * Or when you want to selectively apply an operator\n *\n * ```ts\n * import { interval, take, identity } from 'rxjs';\n *\n * const shouldLimit = () => Math.random() < 0.5;\n *\n * const source$ = interval(1000);\n *\n * const result$ = source$.pipe(shouldLimit() ? take(5) : identity);\n *\n * result$.subscribe({\n * next: console.log\n * });\n * ```\n *\n * @param x Any value that is returned by this function\n * @returns The value passed as the first parameter to this function\n */\nexport function identity(x: T): T {\n return x;\n}\n", "import { identity } from './identity';\nimport { UnaryFunction } from '../types';\n\nexport function pipe(): typeof identity;\nexport function pipe(fn1: UnaryFunction): UnaryFunction;\nexport function pipe(fn1: UnaryFunction, fn2: UnaryFunction): UnaryFunction;\nexport function pipe(fn1: UnaryFunction, fn2: UnaryFunction, fn3: UnaryFunction): UnaryFunction;\nexport function pipe(\n fn1: UnaryFunction,\n fn2: UnaryFunction,\n fn3: UnaryFunction,\n fn4: UnaryFunction\n): UnaryFunction;\nexport function pipe(\n fn1: UnaryFunction,\n fn2: UnaryFunction,\n fn3: UnaryFunction,\n fn4: UnaryFunction,\n fn5: UnaryFunction\n): UnaryFunction;\nexport function pipe(\n fn1: UnaryFunction,\n fn2: UnaryFunction,\n fn3: UnaryFunction,\n fn4: UnaryFunction,\n fn5: UnaryFunction,\n fn6: UnaryFunction\n): UnaryFunction;\nexport function pipe(\n fn1: UnaryFunction,\n fn2: UnaryFunction,\n fn3: UnaryFunction,\n fn4: UnaryFunction,\n fn5: UnaryFunction,\n fn6: UnaryFunction,\n fn7: UnaryFunction\n): UnaryFunction;\nexport function pipe(\n fn1: UnaryFunction,\n fn2: UnaryFunction,\n fn3: UnaryFunction,\n fn4: UnaryFunction,\n fn5: UnaryFunction,\n fn6: UnaryFunction,\n fn7: UnaryFunction,\n fn8: UnaryFunction\n): UnaryFunction;\nexport function pipe(\n fn1: UnaryFunction,\n fn2: UnaryFunction,\n fn3: UnaryFunction,\n fn4: UnaryFunction,\n fn5: UnaryFunction,\n fn6: UnaryFunction,\n fn7: UnaryFunction,\n fn8: UnaryFunction,\n fn9: UnaryFunction\n): UnaryFunction;\nexport function pipe(\n fn1: UnaryFunction,\n fn2: UnaryFunction,\n fn3: UnaryFunction,\n fn4: UnaryFunction,\n fn5: UnaryFunction,\n fn6: UnaryFunction,\n fn7: UnaryFunction,\n fn8: UnaryFunction,\n fn9: UnaryFunction,\n ...fns: UnaryFunction[]\n): UnaryFunction;\n\n/**\n * pipe() can be called on one or more functions, each of which can take one argument (\"UnaryFunction\")\n * and uses it to return a value.\n * It returns a function that takes one argument, passes it to the first UnaryFunction, and then\n * passes the result to the next one, passes that result to the next one, and so on. \n */\nexport function pipe(...fns: Array>): UnaryFunction {\n return pipeFromArray(fns);\n}\n\n/** @internal */\nexport function pipeFromArray(fns: Array>): UnaryFunction {\n if (fns.length === 0) {\n return identity as UnaryFunction;\n }\n\n if (fns.length === 1) {\n return fns[0];\n }\n\n return function piped(input: T): R {\n return fns.reduce((prev: any, fn: UnaryFunction) => fn(prev), input as any);\n };\n}\n", "import { Operator } from './Operator';\nimport { SafeSubscriber, Subscriber } from './Subscriber';\nimport { isSubscription, Subscription } from './Subscription';\nimport { TeardownLogic, OperatorFunction, Subscribable, Observer } from './types';\nimport { observable as Symbol_observable } from './symbol/observable';\nimport { pipeFromArray } from './util/pipe';\nimport { config } from './config';\nimport { isFunction } from './util/isFunction';\nimport { errorContext } from './util/errorContext';\n\n/**\n * A representation of any set of values over any amount of time. This is the most basic building block\n * of RxJS.\n */\nexport class Observable implements Subscribable {\n /**\n * @deprecated Internal implementation detail, do not use directly. Will be made internal in v8.\n */\n source: Observable | undefined;\n\n /**\n * @deprecated Internal implementation detail, do not use directly. Will be made internal in v8.\n */\n operator: Operator | undefined;\n\n /**\n * @param subscribe The function that is called when the Observable is\n * initially subscribed to. This function is given a Subscriber, to which new values\n * can be `next`ed, or an `error` method can be called to raise an error, or\n * `complete` can be called to notify of a successful completion.\n */\n constructor(subscribe?: (this: Observable, subscriber: Subscriber) => TeardownLogic) {\n if (subscribe) {\n this._subscribe = subscribe;\n }\n }\n\n // HACK: Since TypeScript inherits static properties too, we have to\n // fight against TypeScript here so Subject can have a different static create signature\n /**\n * Creates a new Observable by calling the Observable constructor\n * @param subscribe the subscriber function to be passed to the Observable constructor\n * @return A new observable.\n * @deprecated Use `new Observable()` instead. Will be removed in v8.\n */\n static create: (...args: any[]) => any = (subscribe?: (subscriber: Subscriber) => TeardownLogic) => {\n return new Observable(subscribe);\n };\n\n /**\n * Creates a new Observable, with this Observable instance as the source, and the passed\n * operator defined as the new observable's operator.\n * @param operator the operator defining the operation to take on the observable\n * @return A new observable with the Operator applied.\n * @deprecated Internal implementation detail, do not use directly. Will be made internal in v8.\n * If you have implemented an operator using `lift`, it is recommended that you create an\n * operator by simply returning `new Observable()` directly. See \"Creating new operators from\n * scratch\" section here: https://rxjs.dev/guide/operators\n */\n lift(operator?: Operator): Observable {\n const observable = new Observable();\n observable.source = this;\n observable.operator = operator;\n return observable;\n }\n\n subscribe(observerOrNext?: Partial> | ((value: T) => void)): Subscription;\n /** @deprecated Instead of passing separate callback arguments, use an observer argument. Signatures taking separate callback arguments will be removed in v8. Details: https://rxjs.dev/deprecations/subscribe-arguments */\n subscribe(next?: ((value: T) => void) | null, error?: ((error: any) => void) | null, complete?: (() => void) | null): Subscription;\n /**\n * Invokes an execution of an Observable and registers Observer handlers for notifications it will emit.\n *\n * Use it when you have all these Observables, but still nothing is happening.\n *\n * `subscribe` is not a regular operator, but a method that calls Observable's internal `subscribe` function. It\n * might be for example a function that you passed to Observable's constructor, but most of the time it is\n * a library implementation, which defines what will be emitted by an Observable, and when it be will emitted. This means\n * that calling `subscribe` is actually the moment when Observable starts its work, not when it is created, as it is often\n * the thought.\n *\n * Apart from starting the execution of an Observable, this method allows you to listen for values\n * that an Observable emits, as well as for when it completes or errors. You can achieve this in two\n * of the following ways.\n *\n * The first way is creating an object that implements {@link Observer} interface. It should have methods\n * defined by that interface, but note that it should be just a regular JavaScript object, which you can create\n * yourself in any way you want (ES6 class, classic function constructor, object literal etc.). In particular, do\n * not attempt to use any RxJS implementation details to create Observers - you don't need them. Remember also\n * that your object does not have to implement all methods. If you find yourself creating a method that doesn't\n * do anything, you can simply omit it. Note however, if the `error` method is not provided and an error happens,\n * it will be thrown asynchronously. Errors thrown asynchronously cannot be caught using `try`/`catch`. Instead,\n * use the {@link onUnhandledError} configuration option or use a runtime handler (like `window.onerror` or\n * `process.on('error)`) to be notified of unhandled errors. Because of this, it's recommended that you provide\n * an `error` method to avoid missing thrown errors.\n *\n * The second way is to give up on Observer object altogether and simply provide callback functions in place of its methods.\n * This means you can provide three functions as arguments to `subscribe`, where the first function is equivalent\n * of a `next` method, the second of an `error` method and the third of a `complete` method. Just as in case of an Observer,\n * if you do not need to listen for something, you can omit a function by passing `undefined` or `null`,\n * since `subscribe` recognizes these functions by where they were placed in function call. When it comes\n * to the `error` function, as with an Observer, if not provided, errors emitted by an Observable will be thrown asynchronously.\n *\n * You can, however, subscribe with no parameters at all. This may be the case where you're not interested in terminal events\n * and you also handled emissions internally by using operators (e.g. using `tap`).\n *\n * Whichever style of calling `subscribe` you use, in both cases it returns a Subscription object.\n * This object allows you to call `unsubscribe` on it, which in turn will stop the work that an Observable does and will clean\n * up all resources that an Observable used. Note that cancelling a subscription will not call `complete` callback\n * provided to `subscribe` function, which is reserved for a regular completion signal that comes from an Observable.\n *\n * Remember that callbacks provided to `subscribe` are not guaranteed to be called asynchronously.\n * It is an Observable itself that decides when these functions will be called. For example {@link of}\n * by default emits all its values synchronously. Always check documentation for how given Observable\n * will behave when subscribed and if its default behavior can be modified with a `scheduler`.\n *\n * #### Examples\n *\n * Subscribe with an {@link guide/observer Observer}\n *\n * ```ts\n * import { of } from 'rxjs';\n *\n * const sumObserver = {\n * sum: 0,\n * next(value) {\n * console.log('Adding: ' + value);\n * this.sum = this.sum + value;\n * },\n * error() {\n * // We actually could just remove this method,\n * // since we do not really care about errors right now.\n * },\n * complete() {\n * console.log('Sum equals: ' + this.sum);\n * }\n * };\n *\n * of(1, 2, 3) // Synchronously emits 1, 2, 3 and then completes.\n * .subscribe(sumObserver);\n *\n * // Logs:\n * // 'Adding: 1'\n * // 'Adding: 2'\n * // 'Adding: 3'\n * // 'Sum equals: 6'\n * ```\n *\n * Subscribe with functions ({@link deprecations/subscribe-arguments deprecated})\n *\n * ```ts\n * import { of } from 'rxjs'\n *\n * let sum = 0;\n *\n * of(1, 2, 3).subscribe(\n * value => {\n * console.log('Adding: ' + value);\n * sum = sum + value;\n * },\n * undefined,\n * () => console.log('Sum equals: ' + sum)\n * );\n *\n * // Logs:\n * // 'Adding: 1'\n * // 'Adding: 2'\n * // 'Adding: 3'\n * // 'Sum equals: 6'\n * ```\n *\n * Cancel a subscription\n *\n * ```ts\n * import { interval } from 'rxjs';\n *\n * const subscription = interval(1000).subscribe({\n * next(num) {\n * console.log(num)\n * },\n * complete() {\n * // Will not be called, even when cancelling subscription.\n * console.log('completed!');\n * }\n * });\n *\n * setTimeout(() => {\n * subscription.unsubscribe();\n * console.log('unsubscribed!');\n * }, 2500);\n *\n * // Logs:\n * // 0 after 1s\n * // 1 after 2s\n * // 'unsubscribed!' after 2.5s\n * ```\n *\n * @param observerOrNext Either an {@link Observer} with some or all callback methods,\n * or the `next` handler that is called for each value emitted from the subscribed Observable.\n * @param error A handler for a terminal event resulting from an error. If no error handler is provided,\n * the error will be thrown asynchronously as unhandled.\n * @param complete A handler for a terminal event resulting from successful completion.\n * @return A subscription reference to the registered handlers.\n */\n subscribe(\n observerOrNext?: Partial> | ((value: T) => void) | null,\n error?: ((error: any) => void) | null,\n complete?: (() => void) | null\n ): Subscription {\n const subscriber = isSubscriber(observerOrNext) ? observerOrNext : new SafeSubscriber(observerOrNext, error, complete);\n\n errorContext(() => {\n const { operator, source } = this;\n subscriber.add(\n operator\n ? // We're dealing with a subscription in the\n // operator chain to one of our lifted operators.\n operator.call(subscriber, source)\n : source\n ? // If `source` has a value, but `operator` does not, something that\n // had intimate knowledge of our API, like our `Subject`, must have\n // set it. We're going to just call `_subscribe` directly.\n this._subscribe(subscriber)\n : // In all other cases, we're likely wrapping a user-provided initializer\n // function, so we need to catch errors and handle them appropriately.\n this._trySubscribe(subscriber)\n );\n });\n\n return subscriber;\n }\n\n /** @internal */\n protected _trySubscribe(sink: Subscriber): TeardownLogic {\n try {\n return this._subscribe(sink);\n } catch (err) {\n // We don't need to return anything in this case,\n // because it's just going to try to `add()` to a subscription\n // above.\n sink.error(err);\n }\n }\n\n /**\n * Used as a NON-CANCELLABLE means of subscribing to an observable, for use with\n * APIs that expect promises, like `async/await`. You cannot unsubscribe from this.\n *\n * **WARNING**: Only use this with observables you *know* will complete. If the source\n * observable does not complete, you will end up with a promise that is hung up, and\n * potentially all of the state of an async function hanging out in memory. To avoid\n * this situation, look into adding something like {@link timeout}, {@link take},\n * {@link takeWhile}, or {@link takeUntil} amongst others.\n *\n * #### Example\n *\n * ```ts\n * import { interval, take } from 'rxjs';\n *\n * const source$ = interval(1000).pipe(take(4));\n *\n * async function getTotal() {\n * let total = 0;\n *\n * await source$.forEach(value => {\n * total += value;\n * console.log('observable -> ' + value);\n * });\n *\n * return total;\n * }\n *\n * getTotal().then(\n * total => console.log('Total: ' + total)\n * );\n *\n * // Expected:\n * // 'observable -> 0'\n * // 'observable -> 1'\n * // 'observable -> 2'\n * // 'observable -> 3'\n * // 'Total: 6'\n * ```\n *\n * @param next A handler for each value emitted by the observable.\n * @return A promise that either resolves on observable completion or\n * rejects with the handled error.\n */\n forEach(next: (value: T) => void): Promise;\n\n /**\n * @param next a handler for each value emitted by the observable\n * @param promiseCtor a constructor function used to instantiate the Promise\n * @return a promise that either resolves on observable completion or\n * rejects with the handled error\n * @deprecated Passing a Promise constructor will no longer be available\n * in upcoming versions of RxJS. This is because it adds weight to the library, for very\n * little benefit. If you need this functionality, it is recommended that you either\n * polyfill Promise, or you create an adapter to convert the returned native promise\n * to whatever promise implementation you wanted. Will be removed in v8.\n */\n forEach(next: (value: T) => void, promiseCtor: PromiseConstructorLike): Promise;\n\n forEach(next: (value: T) => void, promiseCtor?: PromiseConstructorLike): Promise {\n promiseCtor = getPromiseCtor(promiseCtor);\n\n return new promiseCtor((resolve, reject) => {\n const subscriber = new SafeSubscriber({\n next: (value) => {\n try {\n next(value);\n } catch (err) {\n reject(err);\n subscriber.unsubscribe();\n }\n },\n error: reject,\n complete: resolve,\n });\n this.subscribe(subscriber);\n }) as Promise;\n }\n\n /** @internal */\n protected _subscribe(subscriber: Subscriber): TeardownLogic {\n return this.source?.subscribe(subscriber);\n }\n\n /**\n * An interop point defined by the es7-observable spec https://github.com/zenparsing/es-observable\n * @return This instance of the observable.\n */\n [Symbol_observable]() {\n return this;\n }\n\n /* tslint:disable:max-line-length */\n pipe(): Observable;\n pipe(op1: OperatorFunction): Observable;\n pipe(op1: OperatorFunction, op2: OperatorFunction): Observable;\n pipe(op1: OperatorFunction, op2: OperatorFunction, op3: OperatorFunction): Observable;\n pipe(\n op1: OperatorFunction,\n op2: OperatorFunction,\n op3: OperatorFunction,\n op4: OperatorFunction\n ): Observable;\n pipe(\n op1: OperatorFunction,\n op2: OperatorFunction,\n op3: OperatorFunction,\n op4: OperatorFunction,\n op5: OperatorFunction\n ): Observable;\n pipe(\n op1: OperatorFunction,\n op2: OperatorFunction,\n op3: OperatorFunction,\n op4: OperatorFunction,\n op5: OperatorFunction,\n op6: OperatorFunction\n ): Observable;\n pipe(\n op1: OperatorFunction,\n op2: OperatorFunction,\n op3: OperatorFunction,\n op4: OperatorFunction,\n op5: OperatorFunction,\n op6: OperatorFunction,\n op7: OperatorFunction\n ): Observable;\n pipe(\n op1: OperatorFunction,\n op2: OperatorFunction,\n op3: OperatorFunction,\n op4: OperatorFunction,\n op5: OperatorFunction,\n op6: OperatorFunction,\n op7: OperatorFunction,\n op8: OperatorFunction\n ): Observable;\n pipe(\n op1: OperatorFunction,\n op2: OperatorFunction,\n op3: OperatorFunction,\n op4: OperatorFunction,\n op5: OperatorFunction,\n op6: OperatorFunction,\n op7: OperatorFunction,\n op8: OperatorFunction,\n op9: OperatorFunction\n ): Observable;\n pipe(\n op1: OperatorFunction,\n op2: OperatorFunction,\n op3: OperatorFunction,\n op4: OperatorFunction,\n op5: OperatorFunction,\n op6: OperatorFunction,\n op7: OperatorFunction,\n op8: OperatorFunction,\n op9: OperatorFunction,\n ...operations: OperatorFunction[]\n ): Observable;\n /* tslint:enable:max-line-length */\n\n /**\n * Used to stitch together functional operators into a chain.\n *\n * ## Example\n *\n * ```ts\n * import { interval, filter, map, scan } from 'rxjs';\n *\n * interval(1000)\n * .pipe(\n * filter(x => x % 2 === 0),\n * map(x => x + x),\n * scan((acc, x) => acc + x)\n * )\n * .subscribe(x => console.log(x));\n * ```\n *\n * @return The Observable result of all the operators having been called\n * in the order they were passed in.\n */\n pipe(...operations: OperatorFunction[]): Observable {\n return pipeFromArray(operations)(this);\n }\n\n /* tslint:disable:max-line-length */\n /** @deprecated Replaced with {@link firstValueFrom} and {@link lastValueFrom}. Will be removed in v8. Details: https://rxjs.dev/deprecations/to-promise */\n toPromise(): Promise;\n /** @deprecated Replaced with {@link firstValueFrom} and {@link lastValueFrom}. Will be removed in v8. Details: https://rxjs.dev/deprecations/to-promise */\n toPromise(PromiseCtor: typeof Promise): Promise;\n /** @deprecated Replaced with {@link firstValueFrom} and {@link lastValueFrom}. Will be removed in v8. Details: https://rxjs.dev/deprecations/to-promise */\n toPromise(PromiseCtor: PromiseConstructorLike): Promise;\n /* tslint:enable:max-line-length */\n\n /**\n * Subscribe to this Observable and get a Promise resolving on\n * `complete` with the last emission (if any).\n *\n * **WARNING**: Only use this with observables you *know* will complete. If the source\n * observable does not complete, you will end up with a promise that is hung up, and\n * potentially all of the state of an async function hanging out in memory. To avoid\n * this situation, look into adding something like {@link timeout}, {@link take},\n * {@link takeWhile}, or {@link takeUntil} amongst others.\n *\n * @param [promiseCtor] a constructor function used to instantiate\n * the Promise\n * @return A Promise that resolves with the last value emit, or\n * rejects on an error. If there were no emissions, Promise\n * resolves with undefined.\n * @deprecated Replaced with {@link firstValueFrom} and {@link lastValueFrom}. Will be removed in v8. Details: https://rxjs.dev/deprecations/to-promise\n */\n toPromise(promiseCtor?: PromiseConstructorLike): Promise {\n promiseCtor = getPromiseCtor(promiseCtor);\n\n return new promiseCtor((resolve, reject) => {\n let value: T | undefined;\n this.subscribe(\n (x: T) => (value = x),\n (err: any) => reject(err),\n () => resolve(value)\n );\n }) as Promise;\n }\n}\n\n/**\n * Decides between a passed promise constructor from consuming code,\n * A default configured promise constructor, and the native promise\n * constructor and returns it. If nothing can be found, it will throw\n * an error.\n * @param promiseCtor The optional promise constructor to passed by consuming code\n */\nfunction getPromiseCtor(promiseCtor: PromiseConstructorLike | undefined) {\n return promiseCtor ?? config.Promise ?? Promise;\n}\n\nfunction isObserver(value: any): value is Observer {\n return value && isFunction(value.next) && isFunction(value.error) && isFunction(value.complete);\n}\n\nfunction isSubscriber(value: any): value is Subscriber {\n return (value && value instanceof Subscriber) || (isObserver(value) && isSubscription(value));\n}\n", "import { Observable } from '../Observable';\nimport { Subscriber } from '../Subscriber';\nimport { OperatorFunction } from '../types';\nimport { isFunction } from './isFunction';\n\n/**\n * Used to determine if an object is an Observable with a lift function.\n */\nexport function hasLift(source: any): source is { lift: InstanceType['lift'] } {\n return isFunction(source?.lift);\n}\n\n/**\n * Creates an `OperatorFunction`. Used to define operators throughout the library in a concise way.\n * @param init The logic to connect the liftedSource to the subscriber at the moment of subscription.\n */\nexport function operate(\n init: (liftedSource: Observable, subscriber: Subscriber) => (() => void) | void\n): OperatorFunction {\n return (source: Observable) => {\n if (hasLift(source)) {\n return source.lift(function (this: Subscriber, liftedSource: Observable) {\n try {\n return init(liftedSource, this);\n } catch (err) {\n this.error(err);\n }\n });\n }\n throw new TypeError('Unable to lift unknown Observable type');\n };\n}\n", "import { Subscriber } from '../Subscriber';\n\n/**\n * Creates an instance of an `OperatorSubscriber`.\n * @param destination The downstream subscriber.\n * @param onNext Handles next values, only called if this subscriber is not stopped or closed. Any\n * error that occurs in this function is caught and sent to the `error` method of this subscriber.\n * @param onError Handles errors from the subscription, any errors that occur in this handler are caught\n * and send to the `destination` error handler.\n * @param onComplete Handles completion notification from the subscription. Any errors that occur in\n * this handler are sent to the `destination` error handler.\n * @param onFinalize Additional teardown logic here. This will only be called on teardown if the\n * subscriber itself is not already closed. This is called after all other teardown logic is executed.\n */\nexport function createOperatorSubscriber(\n destination: Subscriber,\n onNext?: (value: T) => void,\n onComplete?: () => void,\n onError?: (err: any) => void,\n onFinalize?: () => void\n): Subscriber {\n return new OperatorSubscriber(destination, onNext, onComplete, onError, onFinalize);\n}\n\n/**\n * A generic helper for allowing operators to be created with a Subscriber and\n * use closures to capture necessary state from the operator function itself.\n */\nexport class OperatorSubscriber extends Subscriber {\n /**\n * Creates an instance of an `OperatorSubscriber`.\n * @param destination The downstream subscriber.\n * @param onNext Handles next values, only called if this subscriber is not stopped or closed. Any\n * error that occurs in this function is caught and sent to the `error` method of this subscriber.\n * @param onError Handles errors from the subscription, any errors that occur in this handler are caught\n * and send to the `destination` error handler.\n * @param onComplete Handles completion notification from the subscription. Any errors that occur in\n * this handler are sent to the `destination` error handler.\n * @param onFinalize Additional finalization logic here. This will only be called on finalization if the\n * subscriber itself is not already closed. This is called after all other finalization logic is executed.\n * @param shouldUnsubscribe An optional check to see if an unsubscribe call should truly unsubscribe.\n * NOTE: This currently **ONLY** exists to support the strange behavior of {@link groupBy}, where unsubscription\n * to the resulting observable does not actually disconnect from the source if there are active subscriptions\n * to any grouped observable. (DO NOT EXPOSE OR USE EXTERNALLY!!!)\n */\n constructor(\n destination: Subscriber,\n onNext?: (value: T) => void,\n onComplete?: () => void,\n onError?: (err: any) => void,\n private onFinalize?: () => void,\n private shouldUnsubscribe?: () => boolean\n ) {\n // It's important - for performance reasons - that all of this class's\n // members are initialized and that they are always initialized in the same\n // order. This will ensure that all OperatorSubscriber instances have the\n // same hidden class in V8. This, in turn, will help keep the number of\n // hidden classes involved in property accesses within the base class as\n // low as possible. If the number of hidden classes involved exceeds four,\n // the property accesses will become megamorphic and performance penalties\n // will be incurred - i.e. inline caches won't be used.\n //\n // The reasons for ensuring all instances have the same hidden class are\n // further discussed in this blog post from Benedikt Meurer:\n // https://benediktmeurer.de/2018/03/23/impact-of-polymorphism-on-component-based-frameworks-like-react/\n super(destination);\n this._next = onNext\n ? function (this: OperatorSubscriber, value: T) {\n try {\n onNext(value);\n } catch (err) {\n destination.error(err);\n }\n }\n : super._next;\n this._error = onError\n ? function (this: OperatorSubscriber, err: any) {\n try {\n onError(err);\n } catch (err) {\n // Send any errors that occur down stream.\n destination.error(err);\n } finally {\n // Ensure finalization.\n this.unsubscribe();\n }\n }\n : super._error;\n this._complete = onComplete\n ? function (this: OperatorSubscriber) {\n try {\n onComplete();\n } catch (err) {\n // Send any errors that occur down stream.\n destination.error(err);\n } finally {\n // Ensure finalization.\n this.unsubscribe();\n }\n }\n : super._complete;\n }\n\n unsubscribe() {\n if (!this.shouldUnsubscribe || this.shouldUnsubscribe()) {\n const { closed } = this;\n super.unsubscribe();\n // Execute additional teardown if we have any and we didn't already do so.\n !closed && this.onFinalize?.();\n }\n }\n}\n", "import { Subscription } from '../Subscription';\n\ninterface AnimationFrameProvider {\n schedule(callback: FrameRequestCallback): Subscription;\n requestAnimationFrame: typeof requestAnimationFrame;\n cancelAnimationFrame: typeof cancelAnimationFrame;\n delegate:\n | {\n requestAnimationFrame: typeof requestAnimationFrame;\n cancelAnimationFrame: typeof cancelAnimationFrame;\n }\n | undefined;\n}\n\nexport const animationFrameProvider: AnimationFrameProvider = {\n // When accessing the delegate, use the variable rather than `this` so that\n // the functions can be called without being bound to the provider.\n schedule(callback) {\n let request = requestAnimationFrame;\n let cancel: typeof cancelAnimationFrame | undefined = cancelAnimationFrame;\n const { delegate } = animationFrameProvider;\n if (delegate) {\n request = delegate.requestAnimationFrame;\n cancel = delegate.cancelAnimationFrame;\n }\n const handle = request((timestamp) => {\n // Clear the cancel function. The request has been fulfilled, so\n // attempting to cancel the request upon unsubscription would be\n // pointless.\n cancel = undefined;\n callback(timestamp);\n });\n return new Subscription(() => cancel?.(handle));\n },\n requestAnimationFrame(...args) {\n const { delegate } = animationFrameProvider;\n return (delegate?.requestAnimationFrame || requestAnimationFrame)(...args);\n },\n cancelAnimationFrame(...args) {\n const { delegate } = animationFrameProvider;\n return (delegate?.cancelAnimationFrame || cancelAnimationFrame)(...args);\n },\n delegate: undefined,\n};\n", "import { createErrorClass } from './createErrorClass';\n\nexport interface ObjectUnsubscribedError extends Error {}\n\nexport interface ObjectUnsubscribedErrorCtor {\n /**\n * @deprecated Internal implementation detail. Do not construct error instances.\n * Cannot be tagged as internal: https://github.com/ReactiveX/rxjs/issues/6269\n */\n new (): ObjectUnsubscribedError;\n}\n\n/**\n * An error thrown when an action is invalid because the object has been\n * unsubscribed.\n *\n * @see {@link Subject}\n * @see {@link BehaviorSubject}\n *\n * @class ObjectUnsubscribedError\n */\nexport const ObjectUnsubscribedError: ObjectUnsubscribedErrorCtor = createErrorClass(\n (_super) =>\n function ObjectUnsubscribedErrorImpl(this: any) {\n _super(this);\n this.name = 'ObjectUnsubscribedError';\n this.message = 'object unsubscribed';\n }\n);\n", "import { Operator } from './Operator';\nimport { Observable } from './Observable';\nimport { Subscriber } from './Subscriber';\nimport { Subscription, EMPTY_SUBSCRIPTION } from './Subscription';\nimport { Observer, SubscriptionLike, TeardownLogic } from './types';\nimport { ObjectUnsubscribedError } from './util/ObjectUnsubscribedError';\nimport { arrRemove } from './util/arrRemove';\nimport { errorContext } from './util/errorContext';\n\n/**\n * A Subject is a special type of Observable that allows values to be\n * multicasted to many Observers. Subjects are like EventEmitters.\n *\n * Every Subject is an Observable and an Observer. You can subscribe to a\n * Subject, and you can call next to feed values as well as error and complete.\n */\nexport class Subject extends Observable implements SubscriptionLike {\n closed = false;\n\n private currentObservers: Observer[] | null = null;\n\n /** @deprecated Internal implementation detail, do not use directly. Will be made internal in v8. */\n observers: Observer[] = [];\n /** @deprecated Internal implementation detail, do not use directly. Will be made internal in v8. */\n isStopped = false;\n /** @deprecated Internal implementation detail, do not use directly. Will be made internal in v8. */\n hasError = false;\n /** @deprecated Internal implementation detail, do not use directly. Will be made internal in v8. */\n thrownError: any = null;\n\n /**\n * Creates a \"subject\" by basically gluing an observer to an observable.\n *\n * @deprecated Recommended you do not use. Will be removed at some point in the future. Plans for replacement still under discussion.\n */\n static create: (...args: any[]) => any = (destination: Observer, source: Observable): AnonymousSubject => {\n return new AnonymousSubject(destination, source);\n };\n\n constructor() {\n // NOTE: This must be here to obscure Observable's constructor.\n super();\n }\n\n /** @deprecated Internal implementation detail, do not use directly. Will be made internal in v8. */\n lift(operator: Operator): Observable {\n const subject = new AnonymousSubject(this, this);\n subject.operator = operator as any;\n return subject as any;\n }\n\n /** @internal */\n protected _throwIfClosed() {\n if (this.closed) {\n throw new ObjectUnsubscribedError();\n }\n }\n\n next(value: T) {\n errorContext(() => {\n this._throwIfClosed();\n if (!this.isStopped) {\n if (!this.currentObservers) {\n this.currentObservers = Array.from(this.observers);\n }\n for (const observer of this.currentObservers) {\n observer.next(value);\n }\n }\n });\n }\n\n error(err: any) {\n errorContext(() => {\n this._throwIfClosed();\n if (!this.isStopped) {\n this.hasError = this.isStopped = true;\n this.thrownError = err;\n const { observers } = this;\n while (observers.length) {\n observers.shift()!.error(err);\n }\n }\n });\n }\n\n complete() {\n errorContext(() => {\n this._throwIfClosed();\n if (!this.isStopped) {\n this.isStopped = true;\n const { observers } = this;\n while (observers.length) {\n observers.shift()!.complete();\n }\n }\n });\n }\n\n unsubscribe() {\n this.isStopped = this.closed = true;\n this.observers = this.currentObservers = null!;\n }\n\n get observed() {\n return this.observers?.length > 0;\n }\n\n /** @internal */\n protected _trySubscribe(subscriber: Subscriber): TeardownLogic {\n this._throwIfClosed();\n return super._trySubscribe(subscriber);\n }\n\n /** @internal */\n protected _subscribe(subscriber: Subscriber): Subscription {\n this._throwIfClosed();\n this._checkFinalizedStatuses(subscriber);\n return this._innerSubscribe(subscriber);\n }\n\n /** @internal */\n protected _innerSubscribe(subscriber: Subscriber) {\n const { hasError, isStopped, observers } = this;\n if (hasError || isStopped) {\n return EMPTY_SUBSCRIPTION;\n }\n this.currentObservers = null;\n observers.push(subscriber);\n return new Subscription(() => {\n this.currentObservers = null;\n arrRemove(observers, subscriber);\n });\n }\n\n /** @internal */\n protected _checkFinalizedStatuses(subscriber: Subscriber) {\n const { hasError, thrownError, isStopped } = this;\n if (hasError) {\n subscriber.error(thrownError);\n } else if (isStopped) {\n subscriber.complete();\n }\n }\n\n /**\n * Creates a new Observable with this Subject as the source. You can do this\n * to create custom Observer-side logic of the Subject and conceal it from\n * code that uses the Observable.\n * @return Observable that this Subject casts to.\n */\n asObservable(): Observable {\n const observable: any = new Observable();\n observable.source = this;\n return observable;\n }\n}\n\nexport class AnonymousSubject extends Subject {\n constructor(\n /** @deprecated Internal implementation detail, do not use directly. Will be made internal in v8. */\n public destination?: Observer,\n source?: Observable\n ) {\n super();\n this.source = source;\n }\n\n next(value: T) {\n this.destination?.next?.(value);\n }\n\n error(err: any) {\n this.destination?.error?.(err);\n }\n\n complete() {\n this.destination?.complete?.();\n }\n\n /** @internal */\n protected _subscribe(subscriber: Subscriber): Subscription {\n return this.source?.subscribe(subscriber) ?? EMPTY_SUBSCRIPTION;\n }\n}\n", "import { Subject } from './Subject';\nimport { Subscriber } from './Subscriber';\nimport { Subscription } from './Subscription';\n\n/**\n * A variant of Subject that requires an initial value and emits its current\n * value whenever it is subscribed to.\n */\nexport class BehaviorSubject extends Subject {\n constructor(private _value: T) {\n super();\n }\n\n get value(): T {\n return this.getValue();\n }\n\n /** @internal */\n protected _subscribe(subscriber: Subscriber): Subscription {\n const subscription = super._subscribe(subscriber);\n !subscription.closed && subscriber.next(this._value);\n return subscription;\n }\n\n getValue(): T {\n const { hasError, thrownError, _value } = this;\n if (hasError) {\n throw thrownError;\n }\n this._throwIfClosed();\n return _value;\n }\n\n next(value: T): void {\n super.next((this._value = value));\n }\n}\n", "import { TimestampProvider } from '../types';\n\ninterface DateTimestampProvider extends TimestampProvider {\n delegate: TimestampProvider | undefined;\n}\n\nexport const dateTimestampProvider: DateTimestampProvider = {\n now() {\n // Use the variable rather than `this` so that the function can be called\n // without being bound to the provider.\n return (dateTimestampProvider.delegate || Date).now();\n },\n delegate: undefined,\n};\n", "import { Subject } from './Subject';\nimport { TimestampProvider } from './types';\nimport { Subscriber } from './Subscriber';\nimport { Subscription } from './Subscription';\nimport { dateTimestampProvider } from './scheduler/dateTimestampProvider';\n\n/**\n * A variant of {@link Subject} that \"replays\" old values to new subscribers by emitting them when they first subscribe.\n *\n * `ReplaySubject` has an internal buffer that will store a specified number of values that it has observed. Like `Subject`,\n * `ReplaySubject` \"observes\" values by having them passed to its `next` method. When it observes a value, it will store that\n * value for a time determined by the configuration of the `ReplaySubject`, as passed to its constructor.\n *\n * When a new subscriber subscribes to the `ReplaySubject` instance, it will synchronously emit all values in its buffer in\n * a First-In-First-Out (FIFO) manner. The `ReplaySubject` will also complete, if it has observed completion; and it will\n * error if it has observed an error.\n *\n * There are two main configuration items to be concerned with:\n *\n * 1. `bufferSize` - This will determine how many items are stored in the buffer, defaults to infinite.\n * 2. `windowTime` - The amount of time to hold a value in the buffer before removing it from the buffer.\n *\n * Both configurations may exist simultaneously. So if you would like to buffer a maximum of 3 values, as long as the values\n * are less than 2 seconds old, you could do so with a `new ReplaySubject(3, 2000)`.\n *\n * ### Differences with BehaviorSubject\n *\n * `BehaviorSubject` is similar to `new ReplaySubject(1)`, with a couple of exceptions:\n *\n * 1. `BehaviorSubject` comes \"primed\" with a single value upon construction.\n * 2. `ReplaySubject` will replay values, even after observing an error, where `BehaviorSubject` will not.\n *\n * @see {@link Subject}\n * @see {@link BehaviorSubject}\n * @see {@link shareReplay}\n */\nexport class ReplaySubject extends Subject {\n private _buffer: (T | number)[] = [];\n private _infiniteTimeWindow = true;\n\n /**\n * @param _bufferSize The size of the buffer to replay on subscription\n * @param _windowTime The amount of time the buffered items will stay buffered\n * @param _timestampProvider An object with a `now()` method that provides the current timestamp. This is used to\n * calculate the amount of time something has been buffered.\n */\n constructor(\n private _bufferSize = Infinity,\n private _windowTime = Infinity,\n private _timestampProvider: TimestampProvider = dateTimestampProvider\n ) {\n super();\n this._infiniteTimeWindow = _windowTime === Infinity;\n this._bufferSize = Math.max(1, _bufferSize);\n this._windowTime = Math.max(1, _windowTime);\n }\n\n next(value: T): void {\n const { isStopped, _buffer, _infiniteTimeWindow, _timestampProvider, _windowTime } = this;\n if (!isStopped) {\n _buffer.push(value);\n !_infiniteTimeWindow && _buffer.push(_timestampProvider.now() + _windowTime);\n }\n this._trimBuffer();\n super.next(value);\n }\n\n /** @internal */\n protected _subscribe(subscriber: Subscriber): Subscription {\n this._throwIfClosed();\n this._trimBuffer();\n\n const subscription = this._innerSubscribe(subscriber);\n\n const { _infiniteTimeWindow, _buffer } = this;\n // We use a copy here, so reentrant code does not mutate our array while we're\n // emitting it to a new subscriber.\n const copy = _buffer.slice();\n for (let i = 0; i < copy.length && !subscriber.closed; i += _infiniteTimeWindow ? 1 : 2) {\n subscriber.next(copy[i] as T);\n }\n\n this._checkFinalizedStatuses(subscriber);\n\n return subscription;\n }\n\n private _trimBuffer() {\n const { _bufferSize, _timestampProvider, _buffer, _infiniteTimeWindow } = this;\n // If we don't have an infinite buffer size, and we're over the length,\n // use splice to truncate the old buffer values off. Note that we have to\n // double the size for instances where we're not using an infinite time window\n // because we're storing the values and the timestamps in the same array.\n const adjustedBufferSize = (_infiniteTimeWindow ? 1 : 2) * _bufferSize;\n _bufferSize < Infinity && adjustedBufferSize < _buffer.length && _buffer.splice(0, _buffer.length - adjustedBufferSize);\n\n // Now, if we're not in an infinite time window, remove all values where the time is\n // older than what is allowed.\n if (!_infiniteTimeWindow) {\n const now = _timestampProvider.now();\n let last = 0;\n // Search the array for the first timestamp that isn't expired and\n // truncate the buffer up to that point.\n for (let i = 1; i < _buffer.length && (_buffer[i] as number) <= now; i += 2) {\n last = i;\n }\n last && _buffer.splice(0, last + 1);\n }\n }\n}\n", "import { Scheduler } from '../Scheduler';\nimport { Subscription } from '../Subscription';\nimport { SchedulerAction } from '../types';\n\n/**\n * A unit of work to be executed in a `scheduler`. An action is typically\n * created from within a {@link SchedulerLike} and an RxJS user does not need to concern\n * themselves about creating and manipulating an Action.\n *\n * ```ts\n * class Action extends Subscription {\n * new (scheduler: Scheduler, work: (state?: T) => void);\n * schedule(state?: T, delay: number = 0): Subscription;\n * }\n * ```\n */\nexport class Action extends Subscription {\n constructor(scheduler: Scheduler, work: (this: SchedulerAction, state?: T) => void) {\n super();\n }\n /**\n * Schedules this action on its parent {@link SchedulerLike} for execution. May be passed\n * some context object, `state`. May happen at some point in the future,\n * according to the `delay` parameter, if specified.\n * @param state Some contextual data that the `work` function uses when called by the\n * Scheduler.\n * @param delay Time to wait before executing the work, where the time unit is implicit\n * and defined by the Scheduler.\n * @return A subscription in order to be able to unsubscribe the scheduled work.\n */\n public schedule(state?: T, delay: number = 0): Subscription {\n return this;\n }\n}\n", "import type { TimerHandle } from './timerHandle';\ntype SetIntervalFunction = (handler: () => void, timeout?: number, ...args: any[]) => TimerHandle;\ntype ClearIntervalFunction = (handle: TimerHandle) => void;\n\ninterface IntervalProvider {\n setInterval: SetIntervalFunction;\n clearInterval: ClearIntervalFunction;\n delegate:\n | {\n setInterval: SetIntervalFunction;\n clearInterval: ClearIntervalFunction;\n }\n | undefined;\n}\n\nexport const intervalProvider: IntervalProvider = {\n // When accessing the delegate, use the variable rather than `this` so that\n // the functions can be called without being bound to the provider.\n setInterval(handler: () => void, timeout?: number, ...args) {\n const { delegate } = intervalProvider;\n if (delegate?.setInterval) {\n return delegate.setInterval(handler, timeout, ...args);\n }\n return setInterval(handler, timeout, ...args);\n },\n clearInterval(handle) {\n const { delegate } = intervalProvider;\n return (delegate?.clearInterval || clearInterval)(handle as any);\n },\n delegate: undefined,\n};\n", "import { Action } from './Action';\nimport { SchedulerAction } from '../types';\nimport { Subscription } from '../Subscription';\nimport { AsyncScheduler } from './AsyncScheduler';\nimport { intervalProvider } from './intervalProvider';\nimport { arrRemove } from '../util/arrRemove';\nimport { TimerHandle } from './timerHandle';\n\nexport class AsyncAction extends Action {\n public id: TimerHandle | undefined;\n public state?: T;\n // @ts-ignore: Property has no initializer and is not definitely assigned\n public delay: number;\n protected pending: boolean = false;\n\n constructor(protected scheduler: AsyncScheduler, protected work: (this: SchedulerAction, state?: T) => void) {\n super(scheduler, work);\n }\n\n public schedule(state?: T, delay: number = 0): Subscription {\n if (this.closed) {\n return this;\n }\n\n // Always replace the current state with the new state.\n this.state = state;\n\n const id = this.id;\n const scheduler = this.scheduler;\n\n //\n // Important implementation note:\n //\n // Actions only execute once by default, unless rescheduled from within the\n // scheduled callback. This allows us to implement single and repeat\n // actions via the same code path, without adding API surface area, as well\n // as mimic traditional recursion but across asynchronous boundaries.\n //\n // However, JS runtimes and timers distinguish between intervals achieved by\n // serial `setTimeout` calls vs. a single `setInterval` call. An interval of\n // serial `setTimeout` calls can be individually delayed, which delays\n // scheduling the next `setTimeout`, and so on. `setInterval` attempts to\n // guarantee the interval callback will be invoked more precisely to the\n // interval period, regardless of load.\n //\n // Therefore, we use `setInterval` to schedule single and repeat actions.\n // If the action reschedules itself with the same delay, the interval is not\n // canceled. If the action doesn't reschedule, or reschedules with a\n // different delay, the interval will be canceled after scheduled callback\n // execution.\n //\n if (id != null) {\n this.id = this.recycleAsyncId(scheduler, id, delay);\n }\n\n // Set the pending flag indicating that this action has been scheduled, or\n // has recursively rescheduled itself.\n this.pending = true;\n\n this.delay = delay;\n // If this action has already an async Id, don't request a new one.\n this.id = this.id ?? this.requestAsyncId(scheduler, this.id, delay);\n\n return this;\n }\n\n protected requestAsyncId(scheduler: AsyncScheduler, _id?: TimerHandle, delay: number = 0): TimerHandle {\n return intervalProvider.setInterval(scheduler.flush.bind(scheduler, this), delay);\n }\n\n protected recycleAsyncId(_scheduler: AsyncScheduler, id?: TimerHandle, delay: number | null = 0): TimerHandle | undefined {\n // If this action is rescheduled with the same delay time, don't clear the interval id.\n if (delay != null && this.delay === delay && this.pending === false) {\n return id;\n }\n // Otherwise, if the action's delay time is different from the current delay,\n // or the action has been rescheduled before it's executed, clear the interval id\n if (id != null) {\n intervalProvider.clearInterval(id);\n }\n\n return undefined;\n }\n\n /**\n * Immediately executes this action and the `work` it contains.\n */\n public execute(state: T, delay: number): any {\n if (this.closed) {\n return new Error('executing a cancelled action');\n }\n\n this.pending = false;\n const error = this._execute(state, delay);\n if (error) {\n return error;\n } else if (this.pending === false && this.id != null) {\n // Dequeue if the action didn't reschedule itself. Don't call\n // unsubscribe(), because the action could reschedule later.\n // For example:\n // ```\n // scheduler.schedule(function doWork(counter) {\n // /* ... I'm a busy worker bee ... */\n // var originalAction = this;\n // /* wait 100ms before rescheduling the action */\n // setTimeout(function () {\n // originalAction.schedule(counter + 1);\n // }, 100);\n // }, 1000);\n // ```\n this.id = this.recycleAsyncId(this.scheduler, this.id, null);\n }\n }\n\n protected _execute(state: T, _delay: number): any {\n let errored: boolean = false;\n let errorValue: any;\n try {\n this.work(state);\n } catch (e) {\n errored = true;\n // HACK: Since code elsewhere is relying on the \"truthiness\" of the\n // return here, we can't have it return \"\" or 0 or false.\n // TODO: Clean this up when we refactor schedulers mid-version-8 or so.\n errorValue = e ? e : new Error('Scheduled action threw falsy error');\n }\n if (errored) {\n this.unsubscribe();\n return errorValue;\n }\n }\n\n unsubscribe() {\n if (!this.closed) {\n const { id, scheduler } = this;\n const { actions } = scheduler;\n\n this.work = this.state = this.scheduler = null!;\n this.pending = false;\n\n arrRemove(actions, this);\n if (id != null) {\n this.id = this.recycleAsyncId(scheduler, id, null);\n }\n\n this.delay = null!;\n super.unsubscribe();\n }\n }\n}\n", "import { Action } from './scheduler/Action';\nimport { Subscription } from './Subscription';\nimport { SchedulerLike, SchedulerAction } from './types';\nimport { dateTimestampProvider } from './scheduler/dateTimestampProvider';\n\n/**\n * An execution context and a data structure to order tasks and schedule their\n * execution. Provides a notion of (potentially virtual) time, through the\n * `now()` getter method.\n *\n * Each unit of work in a Scheduler is called an `Action`.\n *\n * ```ts\n * class Scheduler {\n * now(): number;\n * schedule(work, delay?, state?): Subscription;\n * }\n * ```\n *\n * @deprecated Scheduler is an internal implementation detail of RxJS, and\n * should not be used directly. Rather, create your own class and implement\n * {@link SchedulerLike}. Will be made internal in v8.\n */\nexport class Scheduler implements SchedulerLike {\n public static now: () => number = dateTimestampProvider.now;\n\n constructor(private schedulerActionCtor: typeof Action, now: () => number = Scheduler.now) {\n this.now = now;\n }\n\n /**\n * A getter method that returns a number representing the current time\n * (at the time this function was called) according to the scheduler's own\n * internal clock.\n * @return A number that represents the current time. May or may not\n * have a relation to wall-clock time. May or may not refer to a time unit\n * (e.g. milliseconds).\n */\n public now: () => number;\n\n /**\n * Schedules a function, `work`, for execution. May happen at some point in\n * the future, according to the `delay` parameter, if specified. May be passed\n * some context object, `state`, which will be passed to the `work` function.\n *\n * The given arguments will be processed an stored as an Action object in a\n * queue of actions.\n *\n * @param work A function representing a task, or some unit of work to be\n * executed by the Scheduler.\n * @param delay Time to wait before executing the work, where the time unit is\n * implicit and defined by the Scheduler itself.\n * @param state Some contextual data that the `work` function uses when called\n * by the Scheduler.\n * @return A subscription in order to be able to unsubscribe the scheduled work.\n */\n public schedule(work: (this: SchedulerAction, state?: T) => void, delay: number = 0, state?: T): Subscription {\n return new this.schedulerActionCtor(this, work).schedule(state, delay);\n }\n}\n", "import { Scheduler } from '../Scheduler';\nimport { Action } from './Action';\nimport { AsyncAction } from './AsyncAction';\nimport { TimerHandle } from './timerHandle';\n\nexport class AsyncScheduler extends Scheduler {\n public actions: Array> = [];\n /**\n * A flag to indicate whether the Scheduler is currently executing a batch of\n * queued actions.\n * @internal\n */\n public _active: boolean = false;\n /**\n * An internal ID used to track the latest asynchronous task such as those\n * coming from `setTimeout`, `setInterval`, `requestAnimationFrame`, and\n * others.\n * @internal\n */\n public _scheduled: TimerHandle | undefined;\n\n constructor(SchedulerAction: typeof Action, now: () => number = Scheduler.now) {\n super(SchedulerAction, now);\n }\n\n public flush(action: AsyncAction): void {\n const { actions } = this;\n\n if (this._active) {\n actions.push(action);\n return;\n }\n\n let error: any;\n this._active = true;\n\n do {\n if ((error = action.execute(action.state, action.delay))) {\n break;\n }\n } while ((action = actions.shift()!)); // exhaust the scheduler queue\n\n this._active = false;\n\n if (error) {\n while ((action = actions.shift()!)) {\n action.unsubscribe();\n }\n throw error;\n }\n }\n}\n", "import { AsyncAction } from './AsyncAction';\nimport { AsyncScheduler } from './AsyncScheduler';\n\n/**\n *\n * Async Scheduler\n *\n * Schedule task as if you used setTimeout(task, duration)\n *\n * `async` scheduler schedules tasks asynchronously, by putting them on the JavaScript\n * event loop queue. It is best used to delay tasks in time or to schedule tasks repeating\n * in intervals.\n *\n * If you just want to \"defer\" task, that is to perform it right after currently\n * executing synchronous code ends (commonly achieved by `setTimeout(deferredTask, 0)`),\n * better choice will be the {@link asapScheduler} scheduler.\n *\n * ## Examples\n * Use async scheduler to delay task\n * ```ts\n * import { asyncScheduler } from 'rxjs';\n *\n * const task = () => console.log('it works!');\n *\n * asyncScheduler.schedule(task, 2000);\n *\n * // After 2 seconds logs:\n * // \"it works!\"\n * ```\n *\n * Use async scheduler to repeat task in intervals\n * ```ts\n * import { asyncScheduler } from 'rxjs';\n *\n * function task(state) {\n * console.log(state);\n * this.schedule(state + 1, 1000); // `this` references currently executing Action,\n * // which we reschedule with new state and delay\n * }\n *\n * asyncScheduler.schedule(task, 3000, 0);\n *\n * // Logs:\n * // 0 after 3s\n * // 1 after 4s\n * // 2 after 5s\n * // 3 after 6s\n * ```\n */\n\nexport const asyncScheduler = new AsyncScheduler(AsyncAction);\n\n/**\n * @deprecated Renamed to {@link asyncScheduler}. Will be removed in v8.\n */\nexport const async = asyncScheduler;\n", "import { AsyncAction } from './AsyncAction';\nimport { Subscription } from '../Subscription';\nimport { QueueScheduler } from './QueueScheduler';\nimport { SchedulerAction } from '../types';\nimport { TimerHandle } from './timerHandle';\n\nexport class QueueAction extends AsyncAction {\n constructor(protected scheduler: QueueScheduler, protected work: (this: SchedulerAction, state?: T) => void) {\n super(scheduler, work);\n }\n\n public schedule(state?: T, delay: number = 0): Subscription {\n if (delay > 0) {\n return super.schedule(state, delay);\n }\n this.delay = delay;\n this.state = state;\n this.scheduler.flush(this);\n return this;\n }\n\n public execute(state: T, delay: number): any {\n return delay > 0 || this.closed ? super.execute(state, delay) : this._execute(state, delay);\n }\n\n protected requestAsyncId(scheduler: QueueScheduler, id?: TimerHandle, delay: number = 0): TimerHandle {\n // If delay exists and is greater than 0, or if the delay is null (the\n // action wasn't rescheduled) but was originally scheduled as an async\n // action, then recycle as an async action.\n\n if ((delay != null && delay > 0) || (delay == null && this.delay > 0)) {\n return super.requestAsyncId(scheduler, id, delay);\n }\n\n // Otherwise flush the scheduler starting with this action.\n scheduler.flush(this);\n\n // HACK: In the past, this was returning `void`. However, `void` isn't a valid\n // `TimerHandle`, and generally the return value here isn't really used. So the\n // compromise is to return `0` which is both \"falsy\" and a valid `TimerHandle`,\n // as opposed to refactoring every other instanceo of `requestAsyncId`.\n return 0;\n }\n}\n", "import { AsyncScheduler } from './AsyncScheduler';\n\nexport class QueueScheduler extends AsyncScheduler {\n}\n", "import { QueueAction } from './QueueAction';\nimport { QueueScheduler } from './QueueScheduler';\n\n/**\n *\n * Queue Scheduler\n *\n * Put every next task on a queue, instead of executing it immediately\n *\n * `queue` scheduler, when used with delay, behaves the same as {@link asyncScheduler} scheduler.\n *\n * When used without delay, it schedules given task synchronously - executes it right when\n * it is scheduled. However when called recursively, that is when inside the scheduled task,\n * another task is scheduled with queue scheduler, instead of executing immediately as well,\n * that task will be put on a queue and wait for current one to finish.\n *\n * This means that when you execute task with `queue` scheduler, you are sure it will end\n * before any other task scheduled with that scheduler will start.\n *\n * ## Examples\n * Schedule recursively first, then do something\n * ```ts\n * import { queueScheduler } from 'rxjs';\n *\n * queueScheduler.schedule(() => {\n * queueScheduler.schedule(() => console.log('second')); // will not happen now, but will be put on a queue\n *\n * console.log('first');\n * });\n *\n * // Logs:\n * // \"first\"\n * // \"second\"\n * ```\n *\n * Reschedule itself recursively\n * ```ts\n * import { queueScheduler } from 'rxjs';\n *\n * queueScheduler.schedule(function(state) {\n * if (state !== 0) {\n * console.log('before', state);\n * this.schedule(state - 1); // `this` references currently executing Action,\n * // which we reschedule with new state\n * console.log('after', state);\n * }\n * }, 0, 3);\n *\n * // In scheduler that runs recursively, you would expect:\n * // \"before\", 3\n * // \"before\", 2\n * // \"before\", 1\n * // \"after\", 1\n * // \"after\", 2\n * // \"after\", 3\n *\n * // But with queue it logs:\n * // \"before\", 3\n * // \"after\", 3\n * // \"before\", 2\n * // \"after\", 2\n * // \"before\", 1\n * // \"after\", 1\n * ```\n */\n\nexport const queueScheduler = new QueueScheduler(QueueAction);\n\n/**\n * @deprecated Renamed to {@link queueScheduler}. Will be removed in v8.\n */\nexport const queue = queueScheduler;\n", "import { AsyncAction } from './AsyncAction';\nimport { AnimationFrameScheduler } from './AnimationFrameScheduler';\nimport { SchedulerAction } from '../types';\nimport { animationFrameProvider } from './animationFrameProvider';\nimport { TimerHandle } from './timerHandle';\n\nexport class AnimationFrameAction extends AsyncAction {\n constructor(protected scheduler: AnimationFrameScheduler, protected work: (this: SchedulerAction, state?: T) => void) {\n super(scheduler, work);\n }\n\n protected requestAsyncId(scheduler: AnimationFrameScheduler, id?: TimerHandle, delay: number = 0): TimerHandle {\n // If delay is greater than 0, request as an async action.\n if (delay !== null && delay > 0) {\n return super.requestAsyncId(scheduler, id, delay);\n }\n // Push the action to the end of the scheduler queue.\n scheduler.actions.push(this);\n // If an animation frame has already been requested, don't request another\n // one. If an animation frame hasn't been requested yet, request one. Return\n // the current animation frame request id.\n return scheduler._scheduled || (scheduler._scheduled = animationFrameProvider.requestAnimationFrame(() => scheduler.flush(undefined)));\n }\n\n protected recycleAsyncId(scheduler: AnimationFrameScheduler, id?: TimerHandle, delay: number = 0): TimerHandle | undefined {\n // If delay exists and is greater than 0, or if the delay is null (the\n // action wasn't rescheduled) but was originally scheduled as an async\n // action, then recycle as an async action.\n if (delay != null ? delay > 0 : this.delay > 0) {\n return super.recycleAsyncId(scheduler, id, delay);\n }\n // If the scheduler queue has no remaining actions with the same async id,\n // cancel the requested animation frame and set the scheduled flag to\n // undefined so the next AnimationFrameAction will request its own.\n const { actions } = scheduler;\n if (id != null && id === scheduler._scheduled && actions[actions.length - 1]?.id !== id) {\n animationFrameProvider.cancelAnimationFrame(id as number);\n scheduler._scheduled = undefined;\n }\n // Return undefined so the action knows to request a new async id if it's rescheduled.\n return undefined;\n }\n}\n", "import { AsyncAction } from './AsyncAction';\nimport { AsyncScheduler } from './AsyncScheduler';\n\nexport class AnimationFrameScheduler extends AsyncScheduler {\n public flush(action?: AsyncAction): void {\n this._active = true;\n // The async id that effects a call to flush is stored in _scheduled.\n // Before executing an action, it's necessary to check the action's async\n // id to determine whether it's supposed to be executed in the current\n // flush.\n // Previous implementations of this method used a count to determine this,\n // but that was unsound, as actions that are unsubscribed - i.e. cancelled -\n // are removed from the actions array and that can shift actions that are\n // scheduled to be executed in a subsequent flush into positions at which\n // they are executed within the current flush.\n let flushId;\n if (action) {\n flushId = action.id;\n } else {\n flushId = this._scheduled;\n this._scheduled = undefined;\n }\n\n const { actions } = this;\n let error: any;\n action = action || actions.shift()!;\n\n do {\n if ((error = action.execute(action.state, action.delay))) {\n break;\n }\n } while ((action = actions[0]) && action.id === flushId && actions.shift());\n\n this._active = false;\n\n if (error) {\n while ((action = actions[0]) && action.id === flushId && actions.shift()) {\n action.unsubscribe();\n }\n throw error;\n }\n }\n}\n", "import { AnimationFrameAction } from './AnimationFrameAction';\nimport { AnimationFrameScheduler } from './AnimationFrameScheduler';\n\n/**\n *\n * Animation Frame Scheduler\n *\n * Perform task when `window.requestAnimationFrame` would fire\n *\n * When `animationFrame` scheduler is used with delay, it will fall back to {@link asyncScheduler} scheduler\n * behaviour.\n *\n * Without delay, `animationFrame` scheduler can be used to create smooth browser animations.\n * It makes sure scheduled task will happen just before next browser content repaint,\n * thus performing animations as efficiently as possible.\n *\n * ## Example\n * Schedule div height animation\n * ```ts\n * // html:
    \n * import { animationFrameScheduler } from 'rxjs';\n *\n * const div = document.querySelector('div');\n *\n * animationFrameScheduler.schedule(function(height) {\n * div.style.height = height + \"px\";\n *\n * this.schedule(height + 1); // `this` references currently executing Action,\n * // which we reschedule with new state\n * }, 0, 0);\n *\n * // You will see a div element growing in height\n * ```\n */\n\nexport const animationFrameScheduler = new AnimationFrameScheduler(AnimationFrameAction);\n\n/**\n * @deprecated Renamed to {@link animationFrameScheduler}. Will be removed in v8.\n */\nexport const animationFrame = animationFrameScheduler;\n", "import { Observable } from '../Observable';\nimport { SchedulerLike } from '../types';\n\n/**\n * A simple Observable that emits no items to the Observer and immediately\n * emits a complete notification.\n *\n * Just emits 'complete', and nothing else.\n *\n * ![](empty.png)\n *\n * A simple Observable that only emits the complete notification. It can be used\n * for composing with other Observables, such as in a {@link mergeMap}.\n *\n * ## Examples\n *\n * Log complete notification\n *\n * ```ts\n * import { EMPTY } from 'rxjs';\n *\n * EMPTY.subscribe({\n * next: () => console.log('Next'),\n * complete: () => console.log('Complete!')\n * });\n *\n * // Outputs\n * // Complete!\n * ```\n *\n * Emit the number 7, then complete\n *\n * ```ts\n * import { EMPTY, startWith } from 'rxjs';\n *\n * const result = EMPTY.pipe(startWith(7));\n * result.subscribe(x => console.log(x));\n *\n * // Outputs\n * // 7\n * ```\n *\n * Map and flatten only odd numbers to the sequence `'a'`, `'b'`, `'c'`\n *\n * ```ts\n * import { interval, mergeMap, of, EMPTY } from 'rxjs';\n *\n * const interval$ = interval(1000);\n * const result = interval$.pipe(\n * mergeMap(x => x % 2 === 1 ? of('a', 'b', 'c') : EMPTY),\n * );\n * result.subscribe(x => console.log(x));\n *\n * // Results in the following to the console:\n * // x is equal to the count on the interval, e.g. (0, 1, 2, 3, ...)\n * // x will occur every 1000ms\n * // if x % 2 is equal to 1, print a, b, c (each on its own)\n * // if x % 2 is not equal to 1, nothing will be output\n * ```\n *\n * @see {@link Observable}\n * @see {@link NEVER}\n * @see {@link of}\n * @see {@link throwError}\n */\nexport const EMPTY = new Observable((subscriber) => subscriber.complete());\n\n/**\n * @param scheduler A {@link SchedulerLike} to use for scheduling\n * the emission of the complete notification.\n * @deprecated Replaced with the {@link EMPTY} constant or {@link scheduled} (e.g. `scheduled([], scheduler)`). Will be removed in v8.\n */\nexport function empty(scheduler?: SchedulerLike) {\n return scheduler ? emptyScheduled(scheduler) : EMPTY;\n}\n\nfunction emptyScheduled(scheduler: SchedulerLike) {\n return new Observable((subscriber) => scheduler.schedule(() => subscriber.complete()));\n}\n", "import { SchedulerLike } from '../types';\nimport { isFunction } from './isFunction';\n\nexport function isScheduler(value: any): value is SchedulerLike {\n return value && isFunction(value.schedule);\n}\n", "import { SchedulerLike } from '../types';\nimport { isFunction } from './isFunction';\nimport { isScheduler } from './isScheduler';\n\nfunction last(arr: T[]): T | undefined {\n return arr[arr.length - 1];\n}\n\nexport function popResultSelector(args: any[]): ((...args: unknown[]) => unknown) | undefined {\n return isFunction(last(args)) ? args.pop() : undefined;\n}\n\nexport function popScheduler(args: any[]): SchedulerLike | undefined {\n return isScheduler(last(args)) ? args.pop() : undefined;\n}\n\nexport function popNumber(args: any[], defaultValue: number): number {\n return typeof last(args) === 'number' ? args.pop()! : defaultValue;\n}\n", "export const isArrayLike = ((x: any): x is ArrayLike => x && typeof x.length === 'number' && typeof x !== 'function');", "import { isFunction } from \"./isFunction\";\n\n/**\n * Tests to see if the object is \"thennable\".\n * @param value the object to test\n */\nexport function isPromise(value: any): value is PromiseLike {\n return isFunction(value?.then);\n}\n", "import { InteropObservable } from '../types';\nimport { observable as Symbol_observable } from '../symbol/observable';\nimport { isFunction } from './isFunction';\n\n/** Identifies an input as being Observable (but not necessary an Rx Observable) */\nexport function isInteropObservable(input: any): input is InteropObservable {\n return isFunction(input[Symbol_observable]);\n}\n", "import { isFunction } from './isFunction';\n\nexport function isAsyncIterable(obj: any): obj is AsyncIterable {\n return Symbol.asyncIterator && isFunction(obj?.[Symbol.asyncIterator]);\n}\n", "/**\n * Creates the TypeError to throw if an invalid object is passed to `from` or `scheduled`.\n * @param input The object that was passed.\n */\nexport function createInvalidObservableTypeError(input: any) {\n // TODO: We should create error codes that can be looked up, so this can be less verbose.\n return new TypeError(\n `You provided ${\n input !== null && typeof input === 'object' ? 'an invalid object' : `'${input}'`\n } where a stream was expected. You can provide an Observable, Promise, ReadableStream, Array, AsyncIterable, or Iterable.`\n );\n}\n", "export function getSymbolIterator(): symbol {\n if (typeof Symbol !== 'function' || !Symbol.iterator) {\n return '@@iterator' as any;\n }\n\n return Symbol.iterator;\n}\n\nexport const iterator = getSymbolIterator();\n", "import { iterator as Symbol_iterator } from '../symbol/iterator';\nimport { isFunction } from './isFunction';\n\n/** Identifies an input as being an Iterable */\nexport function isIterable(input: any): input is Iterable {\n return isFunction(input?.[Symbol_iterator]);\n}\n", "import { ReadableStreamLike } from '../types';\nimport { isFunction } from './isFunction';\n\nexport async function* readableStreamLikeToAsyncGenerator(readableStream: ReadableStreamLike): AsyncGenerator {\n const reader = readableStream.getReader();\n try {\n while (true) {\n const { value, done } = await reader.read();\n if (done) {\n return;\n }\n yield value!;\n }\n } finally {\n reader.releaseLock();\n }\n}\n\nexport function isReadableStreamLike(obj: any): obj is ReadableStreamLike {\n // We don't want to use instanceof checks because they would return\n // false for instances from another Realm, like an