updated docs strings and added README.md

2026-03-08 17:59:56 +05:30
parent 0fbf0ca0f0
commit de7d04eb1a
26 changed files with 546 additions and 406 deletions
--- a/mcp_docs/modules/omniread.core.json
+++ b/mcp_docs/modules/omniread.core.json
@@ -2,14 +2,14 @@
  "module": "omniread.core",
  "content": {
    "path": "omniread.core",
-    "docstring": "Core domain contracts for OmniRead.\n\n---\n\n## Summary\n\nThis package defines the **format-agnostic domain layer** of OmniRead.\nIt exposes canonical content models and abstract interfaces that are\nimplemented by format-specific modules (HTML, PDF, etc.).\n\nPublic exports from this package are considered **stable contracts** and\nare safe for downstream consumers to depend on.\n\nSubmodules:\n- content: Canonical content models and enums\n- parser: Abstract parsing contracts\n- scraper: Abstract scraping contracts\n\nFormat-specific behavior must not be introduced at this layer.\n\n---\n\n## Public API\n\n    Content\n    ContentType\n\n---",
+    "docstring": "# Summary\n\nCore domain contracts for OmniRead.\n\nThis package defines the **format-agnostic domain layer** of OmniRead.\nIt exposes canonical content models and abstract interfaces that are\nimplemented by format-specific modules (HTML, PDF, etc.).\n\nPublic exports from this package are considered **stable contracts** and\nare safe for downstream consumers to depend on.\n\nSubmodules:\n\n- `content`: Canonical content models and enums.\n- `parser`: Abstract parsing contracts.\n- `scraper`: Abstract scraping contracts.\n\nFormat-specific behavior must not be introduced at this layer.\n\n---\n\n# Public API\n\n- `Content`\n- `ContentType`\n\n---",
    "objects": {
      "Content": {
        "name": "Content",
        "kind": "class",
        "path": "omniread.core.Content",
        "signature": "<bound method Alias.signature of Alias('Content', 'omniread.core.content.Content')>",
-        "docstring": "Normalized representation of extracted content.\n\nNotes:\n    **Responsibilities:**\n\n        - A `Content` instance represents a raw content payload along with minimal contextual metadata describing its origin and type\n        - This class is the primary exchange format between Scrapers, Parsers, and Downstream consumers",
+        "docstring": "Normalized representation of extracted content.\n\nNotes:\n    **Responsibilities:**\n\n        - A `Content` instance represents a raw content payload along with\n          minimal contextual metadata describing its origin and type.\n        - This class is the primary exchange format between scrapers,\n          parsers, and downstream consumers.",
        "members": {
          "raw": {
            "name": "raw",
@@ -46,7 +46,7 @@
        "kind": "class",
        "path": "omniread.core.ContentType",
        "signature": "<bound method Alias.signature of Alias('ContentType', 'omniread.core.content.ContentType')>",
-        "docstring": "Supported MIME types for extracted content.\n\nNotes:\n    **Guarantees:**\n\n        - This enum represents the declared or inferred media type of the content source\n        - It is primarily used for routing content to the appropriate parser or downstream consumer",
+        "docstring": "Supported MIME types for extracted content.\n\nNotes:\n    **Guarantees:**\n\n        - This enum represents the declared or inferred media type of the\n          content source.\n        - It is primarily used for routing content to the appropriate\n          parser or downstream consumer.",
        "members": {
          "HTML": {
            "name": "HTML",
@@ -83,7 +83,7 @@
        "kind": "class",
        "path": "omniread.core.BaseParser",
        "signature": "<bound method Alias.signature of Alias('BaseParser', 'omniread.core.parser.BaseParser')>",
-        "docstring": "Base interface for all parsers.\n\nNotes:\n    **Guarantees:**\n\n        - A parser is a self-contained object that owns the Content it is responsible for interpreting\n        - Consumers may rely on early validation of content compatibility and type-stable return values from `parse()`\n\n    **Responsibilities:**\n\n        - Implementations must declare supported content types via `supported_types`\n        - Implementations must raise parsing-specific exceptions from `parse()`\n        - Implementations must remain deterministic for a given input",
+        "docstring": "Base interface for all parsers.\n\nNotes:\n    **Guarantees:**\n\n        - A parser is a self-contained object that owns the `Content` it is\n          responsible for interpreting.\n        - Consumers may rely on early validation of content compatibility\n          and type-stable return values from `parse()`.\n\n    **Responsibilities:**\n\n        - Implementations must declare supported content types via `supported_types`.\n        - Implementations must raise parsing-specific exceptions from `parse()`.\n        - Implementations must remain deterministic for a given input.",
        "members": {
          "supported_types": {
            "name": "supported_types",
@@ -104,7 +104,7 @@
            "kind": "function",
            "path": "omniread.core.BaseParser.parse",
            "signature": "<bound method Alias.signature of Alias('parse', 'omniread.core.parser.BaseParser.parse')>",
-            "docstring": "Parse the owned content into structured output.\n\nReturns:\n    T:\n        Parsed, structured representation.\n\nRaises:\n    Exception:\n        Parsing-specific errors as defined by the implementation.\n\nNotes:\n    **Responsibilities:**\n\n        - Implementations must fully consume the provided content and return a deterministic, structured output"
+            "docstring": "Parse the owned content into structured output.\n\nReturns:\n    T:\n        Parsed, structured representation.\n\nRaises:\n    Exception:\n        Parsing-specific errors as defined by the implementation.\n\nNotes:\n    **Responsibilities:**\n\n        - Implementations must fully consume the provided content and\n          return a deterministic, structured output."
          },
          "supports": {
            "name": "supports",
@@ -120,14 +120,14 @@
        "kind": "class",
        "path": "omniread.core.BaseScraper",
        "signature": "<bound method Alias.signature of Alias('BaseScraper', 'omniread.core.scraper.BaseScraper')>",
-        "docstring": "Base interface for all scrapers.\n\nNotes:\n    **Responsibilities:**\n\n        - A scraper is responsible ONLY for fetching raw content (bytes) from a source. It must not interpret or parse it\n        - A scraper is a stateless acquisition component that retrieves raw content from a source and returns it as a `Content` object\n        - Scrapers define how content is obtained, not what the content means\n        - Implementations may vary in transport mechanism, authentication strategy, retry and backoff behavior\n\n    **Constraints:**\n\n        - Implementations must not parse content, modify content semantics, or couple scraping logic to a specific parser",
+        "docstring": "Base interface for all scrapers.\n\nNotes:\n    **Responsibilities:**\n\n        - A scraper is responsible ONLY for fetching raw content (bytes)\n          from a source. It must not interpret or parse it.\n        - A scraper is a stateless acquisition component that retrieves raw\n          content from a source and returns it as a `Content` object.\n        - Scrapers define how content is obtained, not what the content means.\n        - Implementations may vary in transport mechanism, authentication\n          strategy, retry and backoff behavior.\n\n    **Constraints:**\n\n        - Implementations must not parse content, modify content semantics,\n          or couple scraping logic to a specific parser.",
        "members": {
          "fetch": {
            "name": "fetch",
            "kind": "function",
            "path": "omniread.core.BaseScraper.fetch",
            "signature": "<bound method Alias.signature of Alias('fetch', 'omniread.core.scraper.BaseScraper.fetch')>",
-            "docstring": "Fetch raw content from the given source.\n\nArgs:\n    source (str):\n        Location identifier (URL, file path, S3 URI, etc.)\n    metadata (Optional[Mapping[str, Any]], optional):\n        Optional hints for the scraper (headers, auth, etc.)\n\nReturns:\n    Content:\n        Content object containing raw bytes and metadata.\n\nRaises:\n    Exception:\n        Retrieval-specific errors as defined by the implementation.\n\nNotes:\n    **Responsibilities:**\n\n        - Implementations must retrieve the content referenced by `source` and return it as raw bytes wrapped in a `Content` object"
+            "docstring": "Fetch raw content from the given source.\n\nArgs:\n    source (str):\n        Location identifier (URL, file path, S3 URI, etc.).\n\n    metadata (Optional[Mapping[str, Any]], optional):\n        Optional hints for the scraper (headers, auth, etc.).\n\nReturns:\n    Content:\n        Content object containing raw bytes and metadata.\n\nRaises:\n    Exception:\n        Retrieval-specific errors as defined by the implementation.\n\nNotes:\n    **Responsibilities:**\n\n        - Implementations must retrieve the content referenced by `source`\n          and return it as raw bytes wrapped in a `Content` object."
          }
        }
      },
@@ -136,7 +136,7 @@
        "kind": "module",
        "path": "omniread.core.content",
        "signature": null,
-        "docstring": "Canonical content models for OmniRead.\n\n---\n\n## Summary\n\nThis module defines the **format-agnostic content representation** used across\nall parsers and scrapers in OmniRead.\n\nThe models defined here represent *what* was extracted, not *how* it was\nretrieved or parsed. Format-specific behavior and metadata must not alter\nthe semantic meaning of these models.",
+        "docstring": "# Summary\n\nCanonical content models for OmniRead.\n\nThis module defines the **format-agnostic content representation** used across\nall parsers and scrapers in OmniRead.\n\nThe models defined here represent *what* was extracted, not *how* it was\nretrieved or parsed. Format-specific behavior and metadata must not alter\nthe semantic meaning of these models.",
        "members": {
          "Enum": {
            "name": "Enum",
@@ -177,8 +177,8 @@
            "name": "ContentType",
            "kind": "class",
            "path": "omniread.core.content.ContentType",
-            "signature": "<bound method Class.signature of Class('ContentType', 21, 42)>",
-            "docstring": "Supported MIME types for extracted content.\n\nNotes:\n    **Guarantees:**\n\n        - This enum represents the declared or inferred media type of the content source\n        - It is primarily used for routing content to the appropriate parser or downstream consumer",
+            "signature": "<bound method Class.signature of Class('ContentType', 19, 42)>",
+            "docstring": "Supported MIME types for extracted content.\n\nNotes:\n    **Guarantees:**\n\n        - This enum represents the declared or inferred media type of the\n          content source.\n        - It is primarily used for routing content to the appropriate\n          parser or downstream consumer.",
            "members": {
              "HTML": {
                "name": "HTML",
@@ -214,8 +214,8 @@
            "name": "Content",
            "kind": "class",
            "path": "omniread.core.content.Content",
-            "signature": "<bound method Class.signature of Class('Content', 45, 75)>",
-            "docstring": "Normalized representation of extracted content.\n\nNotes:\n    **Responsibilities:**\n\n        - A `Content` instance represents a raw content payload along with minimal contextual metadata describing its origin and type\n        - This class is the primary exchange format between Scrapers, Parsers, and Downstream consumers",
+            "signature": "<bound method Class.signature of Class('Content', 45, 77)>",
+            "docstring": "Normalized representation of extracted content.\n\nNotes:\n    **Responsibilities:**\n\n        - A `Content` instance represents a raw content payload along with\n          minimal contextual metadata describing its origin and type.\n        - This class is the primary exchange format between scrapers,\n          parsers, and downstream consumers.",
            "members": {
              "raw": {
                "name": "raw",
@@ -254,7 +254,7 @@
        "kind": "module",
        "path": "omniread.core.parser",
        "signature": null,
-        "docstring": "Abstract parsing contracts for OmniRead.\n\n---\n\n## Summary\n\nThis module defines the **format-agnostic parser interface** used to transform\nraw content into structured, typed representations.\n\nParsers are responsible for:\n- Interpreting a single `Content` instance\n- Validating compatibility with the content type\n- Producing a structured output suitable for downstream consumers\n\nParsers are not responsible for:\n- Fetching or acquiring content\n- Performing retries or error recovery\n- Managing multiple content sources",
+        "docstring": "# Summary\n\nAbstract parsing contracts for OmniRead.\n\nThis module defines the **format-agnostic parser interface** used to transform\nraw content into structured, typed representations.\n\nParsers are responsible for:\n\n- Interpreting a single `Content` instance\n- Validating compatibility with the content type\n- Producing a structured output suitable for downstream consumers\n\nParsers are not responsible for:\n\n- Fetching or acquiring content\n- Performing retries or error recovery\n- Managing multiple content sources",
        "members": {
          "ABC": {
            "name": "ABC",
@@ -296,7 +296,7 @@
            "kind": "class",
            "path": "omniread.core.parser.Content",
            "signature": "<bound method Alias.signature of Alias('Content', 'omniread.core.content.Content')>",
-            "docstring": "Normalized representation of extracted content.\n\nNotes:\n    **Responsibilities:**\n\n        - A `Content` instance represents a raw content payload along with minimal contextual metadata describing its origin and type\n        - This class is the primary exchange format between Scrapers, Parsers, and Downstream consumers",
+            "docstring": "Normalized representation of extracted content.\n\nNotes:\n    **Responsibilities:**\n\n        - A `Content` instance represents a raw content payload along with\n          minimal contextual metadata describing its origin and type.\n        - This class is the primary exchange format between scrapers,\n          parsers, and downstream consumers.",
            "members": {
              "raw": {
                "name": "raw",
@@ -333,7 +333,7 @@
            "kind": "class",
            "path": "omniread.core.parser.ContentType",
            "signature": "<bound method Alias.signature of Alias('ContentType', 'omniread.core.content.ContentType')>",
-            "docstring": "Supported MIME types for extracted content.\n\nNotes:\n    **Guarantees:**\n\n        - This enum represents the declared or inferred media type of the content source\n        - It is primarily used for routing content to the appropriate parser or downstream consumer",
+            "docstring": "Supported MIME types for extracted content.\n\nNotes:\n    **Guarantees:**\n\n        - This enum represents the declared or inferred media type of the\n          content source.\n        - It is primarily used for routing content to the appropriate\n          parser or downstream consumer.",
            "members": {
              "HTML": {
                "name": "HTML",
@@ -376,8 +376,8 @@
            "name": "BaseParser",
            "kind": "class",
            "path": "omniread.core.parser.BaseParser",
-            "signature": "<bound method Class.signature of Class('BaseParser', 30, 108)>",
-            "docstring": "Base interface for all parsers.\n\nNotes:\n    **Guarantees:**\n\n        - A parser is a self-contained object that owns the Content it is responsible for interpreting\n        - Consumers may rely on early validation of content compatibility and type-stable return values from `parse()`\n\n    **Responsibilities:**\n\n        - Implementations must declare supported content types via `supported_types`\n        - Implementations must raise parsing-specific exceptions from `parse()`\n        - Implementations must remain deterministic for a given input",
+            "signature": "<bound method Class.signature of Class('BaseParser', 30, 111)>",
+            "docstring": "Base interface for all parsers.\n\nNotes:\n    **Guarantees:**\n\n        - A parser is a self-contained object that owns the `Content` it is\n          responsible for interpreting.\n        - Consumers may rely on early validation of content compatibility\n          and type-stable return values from `parse()`.\n\n    **Responsibilities:**\n\n        - Implementations must declare supported content types via `supported_types`.\n        - Implementations must raise parsing-specific exceptions from `parse()`.\n        - Implementations must remain deterministic for a given input.",
            "members": {
              "supported_types": {
                "name": "supported_types",
@@ -397,14 +397,14 @@
                "name": "parse",
                "kind": "function",
                "path": "omniread.core.parser.BaseParser.parse",
-                "signature": "<bound method Function.signature of Function('parse', 73, 91)>",
-                "docstring": "Parse the owned content into structured output.\n\nReturns:\n    T:\n        Parsed, structured representation.\n\nRaises:\n    Exception:\n        Parsing-specific errors as defined by the implementation.\n\nNotes:\n    **Responsibilities:**\n\n        - Implementations must fully consume the provided content and return a deterministic, structured output"
+                "signature": "<bound method Function.signature of Function('parse', 75, 94)>",
+                "docstring": "Parse the owned content into structured output.\n\nReturns:\n    T:\n        Parsed, structured representation.\n\nRaises:\n    Exception:\n        Parsing-specific errors as defined by the implementation.\n\nNotes:\n    **Responsibilities:**\n\n        - Implementations must fully consume the provided content and\n          return a deterministic, structured output."
              },
              "supports": {
                "name": "supports",
                "kind": "function",
                "path": "omniread.core.parser.BaseParser.supports",
-                "signature": "<bound method Function.signature of Function('supports', 93, 108)>",
+                "signature": "<bound method Function.signature of Function('supports', 96, 111)>",
                "docstring": "Check whether this parser supports the content's type.\n\nReturns:\n    bool:\n        True if the content type is supported; False otherwise."
              }
            }
@@ -416,7 +416,7 @@
        "kind": "module",
        "path": "omniread.core.scraper",
        "signature": null,
-        "docstring": "Abstract scraping contracts for OmniRead.\n\n---\n\n## Summary\n\nThis module defines the **format-agnostic scraper interface** responsible for\nacquiring raw content from external sources.\n\nScrapers are responsible for:\n- Locating and retrieving raw content bytes\n- Attaching minimal contextual metadata\n- Returning normalized `Content` objects\n\nScrapers are explicitly NOT responsible for:\n- Parsing or interpreting content\n- Inferring structure or semantics\n- Performing content-type specific processing\n\nAll interpretation must be delegated to parsers.",
+        "docstring": "# Summary\n\nAbstract scraping contracts for OmniRead.\n\nThis module defines the **format-agnostic scraper interface** responsible for\nacquiring raw content from external sources.\n\nScrapers are responsible for:\n\n- Locating and retrieving raw content bytes\n- Attaching minimal contextual metadata\n- Returning normalized `Content` objects\n\nScrapers are explicitly NOT responsible for:\n\n- Parsing or interpreting content\n- Inferring structure or semantics\n- Performing content-type specific processing\n\nAll interpretation must be delegated to parsers.",
        "members": {
          "ABC": {
            "name": "ABC",
@@ -458,7 +458,7 @@
            "kind": "class",
            "path": "omniread.core.scraper.Content",
            "signature": "<bound method Alias.signature of Alias('Content', 'omniread.core.content.Content')>",
-            "docstring": "Normalized representation of extracted content.\n\nNotes:\n    **Responsibilities:**\n\n        - A `Content` instance represents a raw content payload along with minimal contextual metadata describing its origin and type\n        - This class is the primary exchange format between Scrapers, Parsers, and Downstream consumers",
+            "docstring": "Normalized representation of extracted content.\n\nNotes:\n    **Responsibilities:**\n\n        - A `Content` instance represents a raw content payload along with\n          minimal contextual metadata describing its origin and type.\n        - This class is the primary exchange format between scrapers,\n          parsers, and downstream consumers.",
            "members": {
              "raw": {
                "name": "raw",
@@ -494,15 +494,15 @@
            "name": "BaseScraper",
            "kind": "class",
            "path": "omniread.core.scraper.BaseScraper",
-            "signature": "<bound method Class.signature of Class('BaseScraper', 30, 76)>",
-            "docstring": "Base interface for all scrapers.\n\nNotes:\n    **Responsibilities:**\n\n        - A scraper is responsible ONLY for fetching raw content (bytes) from a source. It must not interpret or parse it\n        - A scraper is a stateless acquisition component that retrieves raw content from a source and returns it as a `Content` object\n        - Scrapers define how content is obtained, not what the content means\n        - Implementations may vary in transport mechanism, authentication strategy, retry and backoff behavior\n\n    **Constraints:**\n\n        - Implementations must not parse content, modify content semantics, or couple scraping logic to a specific parser",
+            "signature": "<bound method Class.signature of Class('BaseScraper', 30, 82)>",
+            "docstring": "Base interface for all scrapers.\n\nNotes:\n    **Responsibilities:**\n\n        - A scraper is responsible ONLY for fetching raw content (bytes)\n          from a source. It must not interpret or parse it.\n        - A scraper is a stateless acquisition component that retrieves raw\n          content from a source and returns it as a `Content` object.\n        - Scrapers define how content is obtained, not what the content means.\n        - Implementations may vary in transport mechanism, authentication\n          strategy, retry and backoff behavior.\n\n    **Constraints:**\n\n        - Implementations must not parse content, modify content semantics,\n          or couple scraping logic to a specific parser.",
            "members": {
              "fetch": {
                "name": "fetch",
                "kind": "function",
                "path": "omniread.core.scraper.BaseScraper.fetch",
-                "signature": "<bound method Function.signature of Function('fetch', 47, 76)>",
-                "docstring": "Fetch raw content from the given source.\n\nArgs:\n    source (str):\n        Location identifier (URL, file path, S3 URI, etc.)\n    metadata (Optional[Mapping[str, Any]], optional):\n        Optional hints for the scraper (headers, auth, etc.)\n\nReturns:\n    Content:\n        Content object containing raw bytes and metadata.\n\nRaises:\n    Exception:\n        Retrieval-specific errors as defined by the implementation.\n\nNotes:\n    **Responsibilities:**\n\n        - Implementations must retrieve the content referenced by `source` and return it as raw bytes wrapped in a `Content` object"
+                "signature": "<bound method Function.signature of Function('fetch', 51, 82)>",
+                "docstring": "Fetch raw content from the given source.\n\nArgs:\n    source (str):\n        Location identifier (URL, file path, S3 URI, etc.).\n\n    metadata (Optional[Mapping[str, Any]], optional):\n        Optional hints for the scraper (headers, auth, etc.).\n\nReturns:\n    Content:\n        Content object containing raw bytes and metadata.\n\nRaises:\n    Exception:\n        Retrieval-specific errors as defined by the implementation.\n\nNotes:\n    **Responsibilities:**\n\n        - Implementations must retrieve the content referenced by `source`\n          and return it as raw bytes wrapped in a `Content` object."
              }
            }
          }