Merge main

2026-06-11 20:58:29 +03:00 · 2025-03-17 10:05:11 +00:00
parent 7e360240bf b2238427a0
commit 59b910ec30
229 changed files with 61430 additions and 3147 deletions
--- a/src/auto_archiver/modules/api_db/init.py
+++ b/src/auto_archiver/modules/api_db/init.py
@@ -1 +1 @@
-from .api_db import AAApiDb
+from .api_db import AAApiDb
--- a/src/auto_archiver/modules/api_db/manifest.py
+++ b/src/auto_archiver/modules/api_db/manifest.py
@@ -1,5 +1,5 @@
 {
-    "name": "Auto-Archiver API Database",
+    "name": "Auto Archiver API Database",
    "type": ["database"],
    "entry_point": "api_db::AAApiDb",
    "requires_setup": True,
@@ -11,8 +11,7 @@
            "required": True,
            "help": "API endpoint where calls are made to",
        },
-        "api_token": {"default": None,
-                      "help": "API Bearer token."},
+        "api_token": {"default": None, "help": "API Bearer token."},
        "public": {
            "default": False,
            "type": "bool",
@@ -24,9 +23,9 @@
            "help": "which group of users have access to the archive in case public=false as author",
        },
        "use_api_cache": {
-            "default": True,
+            "default": False,
            "type": "bool",
-            "help": "if False then the API database will be queried prior to any archiving operations and stop if the link has already been archived",
+            "help": "if True then the API database will be queried prior to any archiving operations and stop if the link has already been archived",
        },
        "store_results": {
            "default": True,
@@ -39,7 +38,7 @@
        },
    },
    "description": """
-     Provides integration with the Auto-Archiver API for querying and storing archival data.
+     Provides integration with the Auto Archiver API for querying and storing archival data.

 ### Features
 - **API Integration**: Supports querying for existing archives and submitting results.
@@ -49,6 +48,6 @@
 - **Optional Storage**: Archives results conditionally based on configuration.

 ### Setup
-Requires access to an Auto-Archiver API instance and a valid API token.
+Requires access to an Auto Archiver API instance and a valid API token.
     """,
 }
--- a/src/auto_archiver/modules/api_db/api_db.py
+++ b/src/auto_archiver/modules/api_db/api_db.py
@@ -12,10 +12,11 @@ class AAApiDb(Database):
    """Connects to auto-archiver-api instance"""

    def fetch(self, item: Metadata) -> Union[Metadata, bool]:
-        """ query the database for the existence of this item.
-            Helps avoid re-archiving the same URL multiple times.
+        """query the database for the existence of this item.
+        Helps avoid re-archiving the same URL multiple times.
        """
-        if not self.use_api_cache: return
+        if not self.use_api_cache:
+            return

        params = {"url": item.get_url(), "limit": 15}
        headers = {"Authorization": f"Bearer {self.api_token}", "accept": "application/json"}
@@ -32,22 +33,25 @@ class AAApiDb(Database):

    def done(self, item: Metadata, cached: bool = False) -> None:
        """archival result ready - should be saved to DB"""
-        if not self.store_results: return
+        if not self.store_results:
+            return
        if cached:
            logger.debug(f"skipping saving archive of {item.get_url()} to the AA API because it was cached")
            return
        logger.debug(f"saving archive of {item.get_url()} to the AA API.")

        payload = {
-            'author_id': self.author_id,
-            'url': item.get_url(),
-            'public': self.public,
-            'group_id': self.group_id,
-            'tags': list(self.tags),
-            'result': item.to_json(),
+            "author_id": self.author_id,
+            "url": item.get_url(),
+            "public": self.public,
+            "group_id": self.group_id,
+            "tags": list(self.tags),
+            "result": item.to_json(),
        }
        headers = {"Authorization": f"Bearer {self.api_token}"}
-        response = requests.post(os.path.join(self.api_endpoint, "interop/submit-archive"), json=payload, headers=headers)
+        response = requests.post(
+            os.path.join(self.api_endpoint, "interop/submit-archive"), json=payload, headers=headers
+        )

        if response.status_code == 201:
            logger.success(f"AA API: {response.json()}")
--- a/src/auto_archiver/modules/atlos_db/init.py
+++ b/src/auto_archiver/modules/atlos_db/init.py
@@ -1 +0,0 @@
-from .atlos_db import AtlosDb
--- a/src/auto_archiver/modules/atlos_db/manifest.py
+++ b/src/auto_archiver/modules/atlos_db/manifest.py
@@ -1,38 +0,0 @@
-{
-    "name": "Atlos Database",
-    "type": ["database"],
-    "entry_point": "atlos_db::AtlosDb",
-    "requires_setup": True,
-    "dependencies":
-        {"python": ["loguru",
-                    ""],
-         "bin": [""]},
-    "configs": {
-        "api_token": {
-            "default": None,
-            "help": "An Atlos API token. For more information, see https://docs.atlos.org/technical/api/",
-            "required": True,
-            "type": "str",
-        },
-        "atlos_url": {
-            "default": "https://platform.atlos.org",
-            "help": "The URL of your Atlos instance (e.g., https://platform.atlos.org), without a trailing slash.",
-            "type": "str"
-        },
-    },
-    "description": """
-Handles integration with the Atlos platform for managing archival results.
-
-### Features
- Outputs archival results to the Atlos API for storage and tracking.
- Updates failure status with error details when archiving fails.
- Processes and formats metadata, including ISO formatting for datetime fields.
- Skips processing for items without an Atlos ID.
-
-### Setup
-Required configs:
- atlos_url: Base URL for the Atlos API.
- api_token: Authentication token for API access.
-"""
-,
-}
--- a/src/auto_archiver/modules/atlos_db/atlos_db.py
+++ b/src/auto_archiver/modules/atlos_db/atlos_db.py
@@ -1,66 +0,0 @@
-from typing import Union
-
-import requests
-from loguru import logger
-
-from auto_archiver.core import Database
-from auto_archiver.core import Metadata
-
-
-class AtlosDb(Database):
-    """
-    Outputs results to Atlos
-    """
-
-    def failed(self, item: Metadata, reason: str) -> None:
-        """Update DB accordingly for failure"""
-        # If the item has no Atlos ID, there's nothing for us to do
-        if not item.metadata.get("atlos_id"):
-            logger.info(f"Item {item.get_url()} has no Atlos ID, skipping")
-            return
-
-        requests.post(
-            f"{self.atlos_url}/api/v2/source_material/metadata/{item.metadata['atlos_id']}/auto_archiver",
-            headers={"Authorization": f"Bearer {self.api_token}"},
-            json={"metadata": {"processed": True, "status": "error", "error": reason}},
-        ).raise_for_status()
-        logger.info(
-            f"Stored failure for {item.get_url()} (ID {item.metadata['atlos_id']}) on Atlos: {reason}"
-        )
-
-    def fetch(self, item: Metadata) -> Union[Metadata, bool]:
-        """check and fetch if the given item has been archived already, each
-        database should handle its own caching, and configuration mechanisms"""
-        return False
-
-    def _process_metadata(self, item: Metadata) -> dict:
-        """Process metadata for storage on Atlos. Will convert any datetime
-        objects to ISO format."""
-
-        return {
-            k: v.isoformat() if hasattr(v, "isoformat") else v
-            for k, v in item.metadata.items()
-        }
-
-    def done(self, item: Metadata, cached: bool = False) -> None:
-        """archival result ready - should be saved to DB"""
-
-        if not item.metadata.get("atlos_id"):
-            logger.info(f"Item {item.get_url()} has no Atlos ID, skipping")
-            return
-
-        requests.post(
-            f"{self.atlos_url}/api/v2/source_material/metadata/{item.metadata['atlos_id']}/auto_archiver",
-            headers={"Authorization": f"Bearer {self.api_token}"},
-            json={
-                "metadata": dict(
-                    processed=True,
-                    status="success",
-                    results=self._process_metadata(item),
-                )
-            },
-        ).raise_for_status()
-
-        logger.info(
-            f"Stored success for {item.get_url()} (ID {item.metadata['atlos_id']}) on Atlos"
-        )
--- a/src/auto_archiver/modules/atlos_feeder/init.py
+++ b/src/auto_archiver/modules/atlos_feeder/init.py
@@ -1 +0,0 @@
-from .atlos_feeder import AtlosFeeder
--- a/src/auto_archiver/modules/atlos_feeder/manifest.py
+++ b/src/auto_archiver/modules/atlos_feeder/manifest.py
@@ -1,34 +0,0 @@
-{
-    "name": "Atlos Feeder",
-    "type": ["feeder"],
-    "requires_setup": True,
-    "dependencies": {
-        "python": ["loguru", "requests"],
-    },
-    "configs": {
-        "api_token": {
-            "type": "str",
-            "required": True,
-            "help": "An Atlos API token. For more information, see https://docs.atlos.org/technical/api/",
-        },
-        "atlos_url": {
-            "default": "https://platform.atlos.org",
-            "help": "The URL of your Atlos instance (e.g., https://platform.atlos.org), without a trailing slash.",
-            "type": "str"
-        },
-    },
-    "description": """
-    AtlosFeeder: A feeder module that integrates with the Atlos API to fetch source material URLs for archival.
-
-    ### Features
-    - Connects to the Atlos API to retrieve a list of source material URLs.
-    - Filters source materials based on visibility, processing status, and metadata.
-    - Converts filtered source materials into `Metadata` objects with the relevant `atlos_id` and URL.
-    - Iterates through paginated results using a cursor for efficient API interaction.
-
-    ### Notes
-    - Requires an Atlos API endpoint and a valid API token for authentication.
-    - Ensures only unprocessed, visible, and ready-to-archive URLs are returned.
-    - Handles pagination transparently when retrieving data from the Atlos API.
-    """
-}
--- a/src/auto_archiver/modules/atlos_feeder/atlos_feeder.py
+++ b/src/auto_archiver/modules/atlos_feeder/atlos_feeder.py
@@ -1,42 +0,0 @@
-import requests
-from loguru import logger
-
-from auto_archiver.core import Feeder
-from auto_archiver.core import Metadata
-
-
-class AtlosFeeder(Feeder):
-
-    def __iter__(self) -> Metadata:
-        # Get all the urls from the Atlos API
-        count = 0
-        cursor = None
-        while True:
-            response = requests.get(
-                f"{self.atlos_url}/api/v2/source_material",
-                headers={"Authorization": f"Bearer {self.api_token}"},
-                params={"cursor": cursor},
-            )
-            data = response.json()
-            response.raise_for_status()
-            cursor = data["next"]
-
-            for item in data["results"]:
-                if (
-                    item["source_url"] not in [None, ""]
-                    and (
-                        item["metadata"]
-                        .get("auto_archiver", {})
-                        .get("processed", False)
-                        != True
-                    )
-                    and item["visibility"] == "visible"
-                    and item["status"] not in ["processing", "pending"]
-                ):
-                    yield Metadata().set_url(item["source_url"]).set(
-                        "atlos_id", item["id"]
-                    )
-                    count += 1
-
-            if len(data["results"]) == 0 or cursor is None:
-                break
--- a/src/auto_archiver/modules/atlos_feeder_db_storage/init.py
+++ b/src/auto_archiver/modules/atlos_feeder_db_storage/init.py
@@ -0,0 +1 @@
+from .atlos_feeder_db_storage import AtlosFeederDbStorage
--- a/src/auto_archiver/modules/atlos_feeder_db_storage/manifest.py
+++ b/src/auto_archiver/modules/atlos_feeder_db_storage/manifest.py
@@ -0,0 +1,46 @@
+{
+    "name": "Atlos Feeder Database Storage",
+    "type": ["feeder", "database", "storage"],
+    "entry_point": "atlos_feeder_db_storage::AtlosFeederDbStorage",
+    "requires_setup": True,
+    "dependencies": {
+        "python": ["loguru", "requests"],
+    },
+    "configs": {
+        "api_token": {
+            "type": "str",
+            "required": True,
+            "help": "An Atlos API token. For more information, see https://docs.atlos.org/technical/api/",
+        },
+        "atlos_url": {
+            "default": "https://platform.atlos.org",
+            "help": "The URL of your Atlos instance (e.g., https://platform.atlos.org), without a trailing slash.",
+            "type": "str",
+        },
+    },
+    "description": """
+    A module that integrates with the Atlos API to fetch source material URLs for archival, uplaod extracted media,
+    
+    [Atlos](https://www.atlos.org/) is a visual investigation and archiving platform designed for investigative research, journalism, and open-source intelligence (OSINT). 
+    It helps users organize, analyze, and store media from various sources, making it easier to track and investigate digital evidence.
+    
+    To get started create a new project and obtain an API token from the settings page. You can group event's into Atlos's 'incidents'.
+    Here you can add 'source material' by URLn and the Atlos feeder will fetch these URLs for archival.
+    
+    You can use Atlos only as a 'feeder', however you can also implement the 'database' and 'storage' features to store the media files in Atlos which is recommended.
+    The Auto Archiver will retain the Atlos ID for each item, ensuring that the media and database outputs are uplaoded back into the relevant media item.
+    
+    
+    ### Features
+    - Connects to the Atlos API to retrieve a list of source material URLs.
+    - Iterates through the URLs from all source material items which are unprocessed, visible, and ready to archive.
+    - If the storage option is selected, it will store the media files alongside the original source material item in Atlos.
+    - Is the database option is selected it will output the results to the media item, as well as updating failure status with error details when archiving fails.
+    - Skips Storege/ database upload for items without an Atlos ID - restricting that you must use the Atlos feeder so that it has the Atlos ID to store the results with.
+
+    ### Notes
+    - Requires an Atlos account with a project and a valid API token for authentication.
+    - Ensures only unprocessed, visible, and ready-to-archive URLs are returned.
+    - Feches any media items within an Atlos project, regardless of separation into incidents.
+    """,
+}
--- a/src/auto_archiver/modules/atlos_feeder_db_storage/atlos_feeder_db_storage.py
+++ b/src/auto_archiver/modules/atlos_feeder_db_storage/atlos_feeder_db_storage.py
@@ -0,0 +1,143 @@
+import hashlib
+import os
+from typing import IO, Iterator, Optional, Union
+
+import requests
+from loguru import logger
+
+from auto_archiver.core import Database, Feeder, Media, Metadata, Storage
+from auto_archiver.utils import calculate_file_hash
+
+
+class AtlosFeederDbStorage(Feeder, Database, Storage):
+    def setup(self) -> requests.Session:
+        """create and return a persistent session."""
+        self.session = requests.Session()
+
+    def _get(self, endpoint: str, params: Optional[dict] = None) -> dict:
+        """Wrapper for GET requests to the Atlos API."""
+        url = f"{self.atlos_url}{endpoint}"
+        response = self.session.get(url, headers={"Authorization": f"Bearer {self.api_token}"}, params=params)
+        response.raise_for_status()
+        return response.json()
+
+    def _post(
+        self,
+        endpoint: str,
+        json: Optional[dict] = None,
+        params: Optional[dict] = None,
+        files: Optional[dict] = None,
+    ) -> dict:
+        """Wrapper for POST requests to the Atlos API."""
+        url = f"{self.atlos_url}{endpoint}"
+        response = self.session.post(
+            url,
+            headers={"Authorization": f"Bearer {self.api_token}"},
+            json=json,
+            params=params,
+            files=files,
+        )
+        response.raise_for_status()
+        return response.json()
+
+    # ! Atlos Module - Feeder Methods
+
+    def __iter__(self) -> Iterator[Metadata]:
+        """Iterate over unprocessed, visible source materials from Atlos."""
+        cursor = None
+        while True:
+            data = self._get("/api/v2/source_material", params={"cursor": cursor})
+            cursor = data.get("next")
+            results = data.get("results", [])
+            for item in results:
+                if (
+                    item.get("source_url") not in [None, ""]
+                    and not item.get("metadata", {}).get("auto_archiver", {}).get("processed", False)
+                    and item.get("visibility") == "visible"
+                    and item.get("status") not in ["processing", "pending"]
+                ):
+                    yield Metadata().set_url(item["source_url"]).set("atlos_id", item["id"])
+            if not results or cursor is None:
+                break
+
+    # ! Atlos Module - Database Methods
+
+    def failed(self, item: Metadata, reason: str) -> None:
+        """Mark an item as failed in Atlos, if the ID exists."""
+        atlos_id = item.metadata.get("atlos_id")
+        if not atlos_id:
+            logger.info(f"Item {item.get_url()} has no Atlos ID, skipping")
+            return
+        self._post(
+            f"/api/v2/source_material/metadata/{atlos_id}/auto_archiver",
+            json={"metadata": {"processed": True, "status": "error", "error": reason}},
+        )
+        logger.info(f"Stored failure for {item.get_url()} (ID {atlos_id}) on Atlos: {reason}")
+
+    def fetch(self, item: Metadata) -> Union[Metadata, bool]:
+        """check and fetch if the given item has been archived already, each
+        database should handle its own caching, and configuration mechanisms"""
+        return False
+
+    def _process_metadata(self, item: Metadata) -> dict:
+        """Process metadata for storage on Atlos. Will convert any datetime
+        objects to ISO format."""
+        return {k: v.isoformat() if hasattr(v, "isoformat") else v for k, v in item.metadata.items()}
+
+    def done(self, item: Metadata, cached: bool = False) -> None:
+        """Mark an item as successfully archived in Atlos."""
+        atlos_id = item.metadata.get("atlos_id")
+        if not atlos_id:
+            logger.info(f"Item {item.get_url()} has no Atlos ID, skipping")
+            return
+        self._post(
+            f"/api/v2/source_material/metadata/{atlos_id}/auto_archiver",
+            json={
+                "metadata": {
+                    "processed": True,
+                    "status": "success",
+                    "results": self._process_metadata(item),
+                }
+            },
+        )
+        logger.info(f"Stored success for {item.get_url()} (ID {atlos_id}) on Atlos")
+
+    # ! Atlos Module - Storage Methods
+
+    def get_cdn_url(self, _media: Media) -> str:
+        """Return the base Atlos URL as the CDN URL."""
+        return self.atlos_url
+
+    def upload(self, media: Media, metadata: Optional[Metadata] = None, **_kwargs) -> bool:
+        """Upload a media file to Atlos if it has not been uploaded already."""
+        if metadata is None:
+            logger.error(f"No metadata provided for {media.filename}")
+            return False
+
+        atlos_id = metadata.get("atlos_id")
+        if not atlos_id:
+            logger.error(f"No Atlos ID found in metadata; can't store {media.filename} in Atlos.")
+            return False
+
+        media_hash = calculate_file_hash(media.filename, hash_algo=hashlib.sha256, chunksize=4096)
+
+        # Check whether the media has already been uploaded
+        source_material = self._get(f"/api/v2/source_material/{atlos_id}")["result"]
+        existing_media = [artifact.get("file_hash_sha256") for artifact in source_material.get("artifacts", [])]
+        if media_hash in existing_media:
+            logger.info(f"{media.filename} with SHA256 {media_hash} already uploaded to Atlos")
+            return True
+
+        # Upload the media to the Atlos API
+        with open(media.filename, "rb") as file_obj:
+            self._post(
+                f"/api/v2/source_material/upload/{atlos_id}",
+                params={"title": media.properties},
+                files={"file": (os.path.basename(media.filename), file_obj)},
+            )
+        logger.info(f"Uploaded {media.filename} to Atlos with ID {atlos_id} and title {media.key}")
+        return True
+
+    def uploadf(self, file: IO[bytes], key: str, **kwargs: dict) -> bool:
+        """Upload a file-like object; not implemented."""
+        pass
--- a/src/auto_archiver/modules/atlos_storage/init.py
+++ b/src/auto_archiver/modules/atlos_storage/init.py
@@ -1 +0,0 @@
-from .atlos_storage import AtlosStorage
--- a/src/auto_archiver/modules/atlos_storage/manifest.py
+++ b/src/auto_archiver/modules/atlos_storage/manifest.py
@@ -1,32 +0,0 @@
-{
-    "name": "Atlos Storage",
-    "type": ["storage"],
-    "requires_setup": True,
-    "dependencies": {
-        "python": ["loguru", "boto3"],
-        "bin": []
-    },
-    "description": """
-    Stores media files in a [Atlos](https://www.atlos.org/).
-
-    ### Features
-    - Saves media files to Atlos, organizing them into folders based on the provided path structure.
-
-    ### Notes
-    - Requires setup with Atlos credentials.
-    - Files are uploaded to the specified `root_folder_id` and organized by the `media.key` structure.
-    """,
-    "configs": {
-        "api_token": {
-            "default": None,
-            "help": "An Atlos API token. For more information, see https://docs.atlos.org/technical/api/",
-            "required": True,
-            "type": "str"
-        },
-        "atlos_url": {
-            "default": "https://platform.atlos.org",
-            "help": "The URL of your Atlos instance (e.g., https://platform.atlos.org), without a trailing slash.",
-            "type": "str"
-        },
-    }
-}
--- a/src/auto_archiver/modules/atlos_storage/atlos_storage.py
+++ b/src/auto_archiver/modules/atlos_storage/atlos_storage.py
@@ -1,66 +0,0 @@
-import hashlib
-import os
-from typing import IO, Optional
-
-import requests
-from loguru import logger
-
-from auto_archiver.core import Media, Metadata
-from auto_archiver.core import Storage
-
-
-class AtlosStorage(Storage):
-
-    def get_cdn_url(self, _media: Media) -> str:
-        # It's not always possible to provide an exact URL, because it's
-        # possible that the media once uploaded could have been copied to
-        # another project.
-        return self.atlos_url
-    
-    def _hash(self, media: Media) -> str:
-        # Hash the media file using sha-256. We don't use the existing auto archiver
-        # hash because there's no guarantee that the configuerer is using sha-256, which
-        # is how Atlos hashes files.
-
-        sha256 = hashlib.sha256()
-        with open(media.filename, "rb") as f:
-            while True:
-                buf = f.read(4096)
-                if not buf: break
-                sha256.update(buf)
-        return sha256.hexdigest()
-
-    def upload(self, media: Media, metadata: Optional[Metadata]=None, **_kwargs) -> bool:
-        atlos_id = metadata.get("atlos_id")
-        if atlos_id is None:
-            logger.error(f"No Atlos ID found in metadata; can't store {media.filename} on Atlos")
-            return False
-        
-        media_hash = self._hash(media)
-        
-        # Check whether the media has already been uploaded
-        source_material = requests.get(
-            f"{self.atlos_url}/api/v2/source_material/{atlos_id}",
-            headers={"Authorization": f"Bearer {self.api_token}"},
-        ).json()["result"]
-        existing_media = [x["file_hash_sha256"] for x in source_material.get("artifacts", [])]
-        if media_hash in existing_media:
-            logger.info(f"{media.filename} with SHA256 {media_hash} already uploaded to Atlos")
-            return True
-        
-        # Upload the media to the Atlos API
-        requests.post(
-            f"{self.atlos_url}/api/v2/source_material/upload/{atlos_id}",
-            headers={"Authorization": f"Bearer {self.api_token}"},
-            params={
-                "title": media.properties
-            },
-            files={"file": (os.path.basename(media.filename), open(media.filename, "rb"))},
-        ).raise_for_status()
-
-        logger.info(f"Uploaded {media.filename} to Atlos with ID {atlos_id} and title {media.key}")
-        
-        return True
-
-    # must be implemented even if unused
-    def uploadf(self, file: IO[bytes], key: str, **kwargs: dict) -> bool: pass
--- a/src/auto_archiver/modules/cli_feeder/manifest.py
+++ b/src/auto_archiver/modules/cli_feeder/manifest.py
@@ -0,0 +1,22 @@
+{
+    "name": "Command Line Feeder",
+    "type": ["feeder"],
+    "entry_point": "cli_feeder::CLIFeeder",
+    "requires_setup": False,
+    "configs": {
+        "urls": {
+            "default": None,
+            "help": "URL(s) to archive, either a single URL or a list of urls, should not come from config.yaml",
+        },
+    },
+    "description": """
+The Command Line Feeder is the default enabled feeder for the Auto Archiver. It allows you to pass URLs directly to the orchestrator from the command line 
+without the need to specify any additional configuration or command line arguments:
+
+`auto-archiver --feeder cli_feeder -- "https://example.com/1/,https://example.com/2/"`
+
+You can pass multiple URLs by separating them with a space. The URLs will be processed in the order they are provided.
+
+`auto-archiver --feeder cli_feeder -- https://example.com/1/ https://example.com/2/`
+""",
+}
--- a/src/auto_archiver/modules/cli_feeder/cli_feeder.py
+++ b/src/auto_archiver/modules/cli_feeder/cli_feeder.py
@@ -0,0 +1,22 @@
+from loguru import logger
+
+from auto_archiver.core.feeder import Feeder
+from auto_archiver.core.metadata import Metadata
+
+
+class CLIFeeder(Feeder):
+    def setup(self) -> None:
+        self.urls = self.config["urls"]
+        if not self.urls:
+            raise ValueError(
+                "No URLs provided. Please provide at least one URL via the command line, or set up an alternative feeder. Use --help for more information."
+            )
+
+    def __iter__(self) -> Metadata:
+        urls = self.config["urls"]
+        for url in urls:
+            logger.debug(f"Processing {url}")
+            m = Metadata().set_url(url)
+            yield m
+
+        logger.success(f"Processed {len(urls)} URL(s)")
--- a/src/auto_archiver/modules/console_db/init.py
+++ b/src/auto_archiver/modules/console_db/init.py
@@ -1 +1 @@
-from .console_db import ConsoleDb
+from .console_db import ConsoleDb
--- a/src/auto_archiver/modules/console_db/console_db.py
+++ b/src/auto_archiver/modules/console_db/console_db.py
@@ -6,18 +6,18 @@ from auto_archiver.core import Metadata

 class ConsoleDb(Database):
    """
-        Outputs results to the console
+    Outputs results to the console
    """

    def started(self, item: Metadata) -> None:
-        logger.warning(f"STARTED {item}")
+        logger.info(f"STARTED {item}")

-    def failed(self, item: Metadata, reason:str) -> None:
+    def failed(self, item: Metadata, reason: str) -> None:
        logger.error(f"FAILED {item}: {reason}")

    def aborted(self, item: Metadata) -> None:
        logger.warning(f"ABORTED {item}")

-    def done(self, item: Metadata, cached: bool=False) -> None:
+    def done(self, item: Metadata, cached: bool = False) -> None:
        """archival result ready - should be saved to DB"""
-        logger.success(f"DONE {item}")
+        logger.success(f"DONE {item}")
--- a/src/auto_archiver/modules/csv_db/init.py
+++ b/src/auto_archiver/modules/csv_db/init.py
@@ -1 +1 @@
-from .csv_db import CSVDb
+from .csv_db import CSVDb
--- a/src/auto_archiver/modules/csv_db/manifest.py
+++ b/src/auto_archiver/modules/csv_db/manifest.py
@@ -2,12 +2,11 @@
    "name": "CSV Database",
    "type": ["database"],
    "requires_setup": False,
-    "dependencies": {"python": ["loguru"]
-                              },
-    'entry_point': 'csv_db::CSVDb',
+    "dependencies": {"python": ["loguru"]},
+    "entry_point": "csv_db::CSVDb",
    "configs": {
-            "csv_file": {"default": "db.csv", "help": "CSV file name"}
-        },
+        "csv_file": {"default": "db.csv", "help": "CSV file name to save metadata to"},
+    },
    "description": """
 Handles exporting archival results to a CSV file.

--- a/src/auto_archiver/modules/csv_db/csv_db.py
+++ b/src/auto_archiver/modules/csv_db/csv_db.py
@@ -9,14 +9,15 @@ from auto_archiver.core import Metadata

 class CSVDb(Database):
    """
-        Outputs results to a CSV file
+    Outputs results to a CSV file
    """

-    def done(self, item: Metadata, cached: bool=False) -> None:
+    def done(self, item: Metadata, cached: bool = False) -> None:
        """archival result ready - should be saved to DB"""
        logger.success(f"DONE {item}")
        is_empty = not os.path.isfile(self.csv_file) or os.path.getsize(self.csv_file) == 0
        with open(self.csv_file, "a", encoding="utf-8") as outf:
            writer = DictWriter(outf, fieldnames=asdict(Metadata()))
-            if is_empty: writer.writeheader()
+            if is_empty:
+                writer.writeheader()
            writer.writerow(asdict(item))
--- a/src/auto_archiver/modules/csv_feeder/init.py
+++ b/src/auto_archiver/modules/csv_feeder/init.py
@@ -1 +1 @@
-from .csv_feeder import CSVFeeder
+from .csv_feeder import CSVFeeder
--- a/src/auto_archiver/modules/csv_feeder/manifest.py
+++ b/src/auto_archiver/modules/csv_feeder/manifest.py
@@ -1,27 +1,23 @@
 {
    "name": "CSV Feeder",
    "type": ["feeder"],
-    "requires_setup": False,
-    "dependencies": {
-        "python": ["loguru"],
-        "bin": [""]
-    },
-    'requires_setup': True,
-    'entry_point': "csv_feeder::CSVFeeder",
+    "dependencies": {"python": ["loguru"], "bin": [""]},
+    "requires_setup": True,
+    "entry_point": "csv_feeder::CSVFeeder",
    "configs": {
-            "files": {
-                "default": None,
-                "help": "Path to the input file(s) to read the URLs from, comma separated. \
+        "files": {
+            "default": None,
+            "help": "Path to the input file(s) to read the URLs from, comma separated. \
                        Input files should be formatted with one URL per line",
-                "required": True,
-                "type": "valid_file",
-                "nargs": "+",
-            },
-            "column": {
-                "default": None,
-                "help": "Column number or name to read the URLs from, 0-indexed",
-            }
+            "required": True,
+            "type": "valid_file",
+            "nargs": "+",
        },
+        "column": {
+            "default": None,
+            "help": "Column number or name to read the URLs from, 0-indexed",
+        },
+    },
    "description": """
    Reads URLs from CSV files and feeds them into the archiving process.

@@ -33,5 +29,5 @@
    ### Setup
    - Input files should be formatted with one URL per line, with or without a header row.
    - If you have a header row, you can specify the column number or name to read URLs from using the 'column' config option.
-    """
+    """,
 }
--- a/src/auto_archiver/modules/csv_feeder/csv_feeder.py
+++ b/src/auto_archiver/modules/csv_feeder/csv_feeder.py
@@ -5,11 +5,10 @@ from auto_archiver.core import Feeder
 from auto_archiver.core import Metadata
 from auto_archiver.utils import url_or_none

+
 class CSVFeeder(Feeder):
-
    column = None

-
    def __iter__(self) -> Metadata:
        for file in self.files:
            with open(file, "r") as f:
@@ -20,9 +19,11 @@ class CSVFeeder(Feeder):
                    try:
                        url_column = first_row.index(url_column)
                    except ValueError:
-                        logger.error(f"Column {url_column} not found in header row: {first_row}. Did you set the 'column' config correctly?")
+                        logger.error(
+                            f"Column {url_column} not found in header row: {first_row}. Did you set the 'column' config correctly?"
+                        )
                        return
-                elif not(url_or_none(first_row[url_column])):
+                elif not (url_or_none(first_row[url_column])):
                    # it's a header row, but we've been given a column number already
                    logger.debug(f"Skipping header row: {first_row}")
                else:
@@ -35,4 +36,4 @@ class CSVFeeder(Feeder):
                        continue
                    url = row[url_column]
                    logger.debug(f"Processing {url}")
-                    yield Metadata().set_url(url)
+                    yield Metadata().set_url(url)
--- a/src/auto_archiver/modules/gdrive_storage/init.py
+++ b/src/auto_archiver/modules/gdrive_storage/init.py
@@ -1 +1 @@
-from .gdrive_storage import GDriveStorage
+from .gdrive_storage import GDriveStorage
--- a/src/auto_archiver/modules/gdrive_storage/manifest.py
+++ b/src/auto_archiver/modules/gdrive_storage/manifest.py
@@ -19,14 +19,21 @@
        },
        "filename_generator": {
            "default": "static",
-            "help": "how to name stored files: 'random' creates a random string; 'static' uses a replicable strategy such as a hash.",
+            "help": "how to name stored files: 'random' creates a random string; 'static' uses a hash, with the settings of the 'hash_enricher' module (defaults to SHA256 if not enabled).",
            "choices": ["random", "static"],
        },
-        "root_folder_id": {"required": True,
-                           "help": "root google drive folder ID to use as storage, found in URL: 'https://drive.google.com/drive/folders/FOLDER_ID'"},
-        "oauth_token": {"default": None,
-                        "help": "JSON filename with Google Drive OAuth token: check auto-archiver repository scripts folder for create_update_gdrive_oauth_token.py. NOTE: storage used will count towards owner of GDrive folder, therefore it is best to use oauth_token_filename over service_account."},
-        "service_account": {"default": "secrets/service_account.json", "help": "service account JSON file path, same as used for Google Sheets. NOTE: storage used will count towards the developer account."},
+        "root_folder_id": {
+            "required": True,
+            "help": "root google drive folder ID to use as storage, found in URL: 'https://drive.google.com/drive/folders/FOLDER_ID'",
+        },
+        "oauth_token": {
+            "default": None,
+            "help": "JSON filename with Google Drive OAuth token: check auto-archiver repository scripts folder for create_update_gdrive_oauth_token.py. NOTE: storage used will count towards owner of GDrive folder, therefore it is best to use oauth_token_filename over service_account.",
+        },
+        "service_account": {
+            "default": "secrets/service_account.json",
+            "help": "service account JSON file path, same as used for Google Sheets. NOTE: storage used will count towards the developer account.",
+        },
    },
    "description": """
    
@@ -94,5 +101,5 @@ This module integrates Google Drive as a storage backend, enabling automatic fol
    https://davemateer.com/2022/04/28/google-drive-with-python#tokens
    
    
-"""
+""",
 }
--- a/src/auto_archiver/modules/gdrive_storage/gdrive_storage.py
+++ b/src/auto_archiver/modules/gdrive_storage/gdrive_storage.py
@@ -1,4 +1,3 @@
-
 import json
 import os
 import time
@@ -15,12 +14,9 @@ from auto_archiver.core import Media
 from auto_archiver.core import Storage


-
-
 class GDriveStorage(Storage):
-
    def setup(self) -> None:
-        self.scopes = ['https://www.googleapis.com/auth/drive']
+        self.scopes = ["https://www.googleapis.com/auth/drive"]
        # Initialize Google Drive service
        self._setup_google_drive_service()

@@ -37,25 +33,25 @@ class GDriveStorage(Storage):

    def _initialize_with_oauth_token(self):
        """Initialize Google Drive service with OAuth token."""
-        with open(self.oauth_token, 'r') as stream:
+        with open(self.oauth_token, "r") as stream:
            creds_json = json.load(stream)
-            creds_json['refresh_token'] = creds_json.get("refresh_token", "")
+            creds_json["refresh_token"] = creds_json.get("refresh_token", "")

        creds = Credentials.from_authorized_user_info(creds_json, self.scopes)
        if not creds.valid and creds.expired and creds.refresh_token:
            creds.refresh(Request())
-            with open(self.oauth_token, 'w') as token_file:
+            with open(self.oauth_token, "w") as token_file:
                logger.debug("Saving refreshed OAuth token.")
                token_file.write(creds.to_json())
        elif not creds.valid:
            raise ValueError("Invalid OAuth token. Please regenerate the token.")

-        return build('drive', 'v3', credentials=creds)
+        return build("drive", "v3", credentials=creds)

    def _initialize_with_service_account(self):
        """Initialize Google Drive service with service account."""
        creds = service_account.Credentials.from_service_account_file(self.service_account, scopes=self.scopes)
-        return build('drive', 'v3', credentials=creds)
+        return build("drive", "v3", credentials=creds)

    def get_cdn_url(self, media: Media) -> str:
        """
@@ -79,7 +75,7 @@ class GDriveStorage(Storage):
        return f"https://drive.google.com/file/d/{file_id}/view?usp=sharing"

    def upload(self, media: Media, **kwargs) -> bool:
-        logger.debug(f'[{self.__class__.__name__}] storing file {media.filename} with key {media.key}')
+        logger.debug(f"[{self.__class__.__name__}] storing file {media.filename} with key {media.key}")
        """
        1. for each sub-folder in the path check if exists or create
        2. upload file to root_id/other_paths.../filename
@@ -95,25 +91,30 @@ class GDriveStorage(Storage):
            parent_id = upload_to

        # upload file to gd
-        logger.debug(f'uploading {filename=} to folder id {upload_to}')
-        file_metadata = {
-            'name': [filename],
-            'parents': [upload_to]
-        }
+        logger.debug(f"uploading {filename=} to folder id {upload_to}")
+        file_metadata = {"name": [filename], "parents": [upload_to]}
        media = MediaFileUpload(media.filename, resumable=True)
-        gd_file = self.service.files().create(supportsAllDrives=True, body=file_metadata, media_body=media, fields='id').execute()
-        logger.debug(f'uploadf: uploaded file {gd_file["id"]} successfully in folder={upload_to}')
+        gd_file = (
+            self.service.files()
+            .create(supportsAllDrives=True, body=file_metadata, media_body=media, fields="id")
+            .execute()
+        )
+        logger.debug(f"uploadf: uploaded file {gd_file['id']} successfully in folder={upload_to}")

    # must be implemented even if unused
-    def uploadf(self, file: IO[bytes], key: str, **kwargs: dict) -> bool: pass
+    def uploadf(self, file: IO[bytes], key: str, **kwargs: dict) -> bool:
+        pass

-    def _get_id_from_parent_and_name(self, parent_id: str,
-                                     name: str,
-                                     retries: int = 1,
-                                     sleep_seconds: int = 10,
-                                     use_mime_type: bool = False,
-                                     raise_on_missing: bool = True,
-                                     use_cache=False):
+    def _get_id_from_parent_and_name(
+        self,
+        parent_id: str,
+        name: str,
+        retries: int = 1,
+        sleep_seconds: int = 10,
+        use_mime_type: bool = False,
+        raise_on_missing: bool = True,
+        use_cache=False,
+    ):
        """
        Retrieves the id of a folder or file from its @name and the @parent_id folder
        Optionally does multiple @retries and sleeps @sleep_seconds between them
@@ -134,32 +135,39 @@ class GDriveStorage(Storage):
        debug_header: str = f"[searching {name=} in {parent_id=}]"
        query_string = f"'{parent_id}' in parents and name = '{name}' and trashed = false "
        if use_mime_type:
-            query_string += f" and mimeType='application/vnd.google-apps.folder' "
+            query_string += " and mimeType='application/vnd.google-apps.folder' "

        for attempt in range(retries):
-            results = self.service.files().list(
-                # both below for Google Shared Drives
-                supportsAllDrives=True,
-                includeItemsFromAllDrives=True,
-                q=query_string,
-                spaces='drive',  # ie not appDataFolder or photos
-                fields='files(id, name)'
-            ).execute()
-            items = results.get('files', [])
+            results = (
+                self.service.files()
+                .list(
+                    # both below for Google Shared Drives
+                    supportsAllDrives=True,
+                    includeItemsFromAllDrives=True,
+                    q=query_string,
+                    spaces="drive",  # ie not appDataFolder or photos
+                    fields="files(id, name)",
+                )
+                .execute()
+            )
+            items = results.get("files", [])

            if len(items) > 0:
-                logger.debug(f"{debug_header} found {len(items)} matches, returning last of {','.join([i['id'] for i in items])}")
-                _id = items[-1]['id']
-                if use_cache: self.api_cache[cache_key] = _id
+                logger.debug(
+                    f"{debug_header} found {len(items)} matches, returning last of {','.join([i['id'] for i in items])}"
+                )
+                _id = items[-1]["id"]
+                if use_cache:
+                    self.api_cache[cache_key] = _id
                return _id
            else:
-                logger.debug(f'{debug_header} not found, attempt {attempt+1}/{retries}.')
+                logger.debug(f"{debug_header} not found, attempt {attempt + 1}/{retries}.")
                if attempt < retries - 1:
-                    logger.debug(f'sleeping for {sleep_seconds} second(s)')
+                    logger.debug(f"sleeping for {sleep_seconds} second(s)")
                    time.sleep(sleep_seconds)

        if raise_on_missing:
-            raise ValueError(f'{debug_header} not found after {retries} attempt(s)')
+            raise ValueError(f"{debug_header} not found after {retries} attempt(s)")
        return None

    def _mkdir(self, name: str, parent_id: str):
@@ -167,12 +175,7 @@ class GDriveStorage(Storage):
        Creates a new GDrive folder @name inside folder @parent_id
        Returns id of the created folder
        """
-        logger.debug(f'Creating new folder with {name=} inside {parent_id=}')
-        file_metadata = {
-            'name': [name],
-            'mimeType': 'application/vnd.google-apps.folder',
-            'parents': [parent_id]
-        }
-        gd_folder = self.service.files().create(supportsAllDrives=True, body=file_metadata, fields='id').execute()
-        return gd_folder.get('id')
-
+        logger.debug(f"Creating new folder with {name=} inside {parent_id=}")
+        file_metadata = {"name": [name], "mimeType": "application/vnd.google-apps.folder", "parents": [parent_id]}
+        gd_folder = self.service.files().create(supportsAllDrives=True, body=file_metadata, fields="id").execute()
+        return gd_folder.get("id")
--- a/src/auto_archiver/modules/generic_extractor/init.py
+++ b/src/auto_archiver/modules/generic_extractor/init.py
@@ -1 +1 @@
-from .generic_extractor import GenericExtractor
+from .generic_extractor import GenericExtractor
--- a/src/auto_archiver/modules/generic_extractor/manifest.py
+++ b/src/auto_archiver/modules/generic_extractor/manifest.py
@@ -28,6 +28,13 @@ the broader archiving framework.
 metadata objects. Some dropins are included in this generic_archiver by default, but
 custom dropins can be created to handle additional websites and passed to the archiver
 via the command line using the `--dropins` option (TODO!).
+
+### Auto-Updates
+
+The Generic Extractor will also automatically check for updates to `yt-dlp` (every 5 days by default).
+This can be configured using the `ytdlp_update_interval` setting (or disabled by setting it to -1).
+If you are having issues with the extractor, you can review the version of `yt-dlp` being used with `yt-dlp --version`.
+
 """,
    "configs": {
        "subtitles": {"default": True, "help": "download subtitles if available", "type": "bool"},
@@ -64,5 +71,17 @@ via the command line using the `--dropins` option (TODO!).
            "default": "inf",
            "help": "Use to limit the number of videos to download when a channel or long page is being extracted. 'inf' means no limit.",
        },
+        "ytdlp_update_interval": {
+            "default": 5,
+            "help": "How often to check for yt-dlp updates (days). If positive, will check and update yt-dlp every [num] days. Set it to -1 to disable, or 0 to always update on every run.",
+            "type": "int",
+        },
+        "ytdlp_args": {
+            "default": "",
+            "help": "Additional arguments to pass to yt-dlp, e.g. --no-check-certificate or --plugin-dirs.\
+See yt-dlp documentation here for more information: https://github.com/yt-dlp/yt-dlp?tab=readme-ov-file#general-options\
+Note: this is not to be confused with 'extractor_args' which are specific to the extractor itself.",
+            "type": "str",
+        },
    },
 }
--- a/src/auto_archiver/modules/generic_extractor/bluesky.py
+++ b/src/auto_archiver/modules/generic_extractor/bluesky.py
@@ -4,15 +4,16 @@ from auto_archiver.core.extractor import Extractor
 from auto_archiver.core.metadata import Metadata, Media
 from .dropin import GenericDropin, InfoExtractor

-class Bluesky(GenericDropin):

+class Bluesky(GenericDropin):
    def create_metadata(self, post: dict, ie_instance: InfoExtractor, archiver: Extractor, url: str) -> Metadata:
        result = Metadata()
        result.set_url(url)
        result.set_title(post["record"]["text"])
        result.set_timestamp(post["record"]["createdAt"])
        for k, v in self._get_post_data(post).items():
-            if v: result.set(k, v)
+            if v:
+                result.set(k, v)

        # download if embeds present (1 video XOR >=1 images)
        for media in self._download_bsky_embeds(post, archiver):
@@ -23,12 +24,12 @@ class Bluesky(GenericDropin):

    def extract_post(self, url: str, ie_instance: InfoExtractor) -> dict:
        # TODO: If/when this PR (https://github.com/yt-dlp/yt-dlp/pull/12098) is merged on ytdlp, remove the comments and delete the code below
-        handle, video_id = ie_instance._match_valid_url(url).group('handle', 'id')
+        handle, video_id = ie_instance._match_valid_url(url).group("handle", "id")
        return ie_instance._extract_post(handle=handle, post_id=video_id)

    def _download_bsky_embeds(self, post: dict, archiver: Extractor) -> list[Media]:
        """
-        Iterates over image(s) or video in a Bluesky post and downloads them        
+        Iterates over image(s) or video in a Bluesky post and downloads them
        """
        media = []
        embed = post.get("record", {}).get("embed", {})
@@ -37,16 +38,15 @@ class Bluesky(GenericDropin):

        media_url = "https://bsky.social/xrpc/com.atproto.sync.getBlob?cid={}&did={}"
        for image_media in image_medias:
-            url = media_url.format(image_media['image']['ref']['$link'], post['author']['did'])
+            url = media_url.format(image_media["image"]["ref"]["$link"], post["author"]["did"])
            image_media = archiver.download_from_url(url)
            media.append(Media(image_media))
        for video_media in video_medias:
-            url = media_url.format(video_media['ref']['$link'], post['author']['did'])
+            url = media_url.format(video_media["ref"]["$link"], post["author"]["did"])
            video_media = archiver.download_from_url(url)
            media.append(Media(video_media))
        return media

-
    def _get_post_data(self, post: dict) -> dict:
        """
        Extracts relevant information returned by the .getPostThread api call (excluding text/created_at): author, mentions, tags, links.
@@ -74,4 +74,4 @@ class Bluesky(GenericDropin):
            res["tags"] = tags
        if links:
            res["links"] = links
-        return res
+        return res
--- a/src/auto_archiver/modules/generic_extractor/dropin.py
+++ b/src/auto_archiver/modules/generic_extractor/dropin.py
@@ -3,11 +3,12 @@ from yt_dlp.extractor.common import InfoExtractor
 from auto_archiver.core.metadata import Metadata
 from auto_archiver.core.extractor import Extractor

+
 class GenericDropin:
    """Base class for dropins for the generic extractor.
-    
+
    In many instances, an extractor will exist in ytdlp, but it will only process videos.
-    Dropins can be created and used to make use of the already-written private code of a 
+    Dropins can be created and used to make use of the already-written private code of a
    specific extractor from ytdlp.

    The dropin should be able to handle the following methods:
@@ -31,21 +32,19 @@ class GenericDropin:
        This method should return the post data from the url.
        """
        raise NotImplementedError("This method should be implemented in the subclass")
-    

    def create_metadata(self, post: dict, ie_instance: InfoExtractor, archiver: Extractor, url: str) -> Metadata:
        """
        This method should create a Metadata object from the post data.
        """
        raise NotImplementedError("This method should be implemented in the subclass")
-    

    def skip_ytdlp_download(self, url: str, ie_instance: InfoExtractor):
        """
        This method should return True if you want to skip the ytdlp download method.
        """
        return False
-    
+
    def keys_to_clean(self, video_data: dict, info_extractor: InfoExtractor):
        """
        This method should return a list of strings (keys) to clean from the video_data dict.
@@ -53,16 +52,16 @@ class GenericDropin:
        E.g. ["uploader", "uploader_id", "tiktok_specific_field"]
        """
        return []
-    
+
    def download_additional_media(self, video_data: dict, info_extractor: InfoExtractor, metadata: Metadata):
        """
        This method should download any additional media from the post.
        """
        return metadata
-    
+
    def is_suitable(self, url, info_extractor: InfoExtractor):
        """
        Used to override the InfoExtractor's 'is_suitable' method. Dropins should override this method to return True if the url is suitable for the extractor
        (based on being able to parse other URLs)
        """
-        return False
+        return False
--- a/src/auto_archiver/modules/generic_extractor/facebook.py
+++ b/src/auto_archiver/modules/generic_extractor/facebook.py
@@ -1,7 +1,6 @@
 import re
 from .dropin import GenericDropin
 from auto_archiver.core.metadata import Metadata
-from auto_archiver.core.media import Media

 # TODO: Remove if / when  https://github.com/yt-dlp/yt-dlp/pull/12275 is merged
 from yt_dlp.utils import (
@@ -12,77 +11,124 @@ from yt_dlp.utils import (
    merge_dicts,
    int_or_none,
    parse_count,
-
 )

+
 def _extract_metadata(self, webpage, video_id):
-    post_data = [self._parse_json(j, video_id, fatal=False) for j in re.findall(
-        r'data-sjs>({.*?ScheduledServerJS.*?})</script>', webpage)]
-    post = traverse_obj(post_data, (
-        ..., 'require', ..., ..., ..., '__bbox', 'require', ..., ..., ..., '__bbox', 'result', 'data'), expected_type=dict) or []
-    media = traverse_obj(post, (..., 'attachments', ..., lambda k, v: (
-        k == 'media' and str(v['id']) == video_id and v['__typename'] == 'Video')), expected_type=dict)
-    title = get_first(media, ('title', 'text'))
-    description = get_first(media, ('creation_story', 'comet_sections', 'message', 'story', 'message', 'text'))
-    page_title = title or self._html_search_regex((
-        r'<h2\s+[^>]*class="uiHeaderTitle"[^>]*>(?P<content>[^<]*)</h2>',
-        r'(?s)<span class="fbPhotosPhotoCaption".*?id="fbPhotoPageCaption"><span class="hasCaption">(?P<content>.*?)</span>',
-        self._meta_regex('og:title'), self._meta_regex('twitter:title'), r'<title>(?P<content>.+?)</title>',
-    ), webpage, 'title', default=None, group='content')
+    post_data = [
+        self._parse_json(j, video_id, fatal=False)
+        for j in re.findall(r"data-sjs>({.*?ScheduledServerJS.*?})</script>", webpage)
+    ]
+    post = (
+        traverse_obj(
+            post_data,
+            (..., "require", ..., ..., ..., "__bbox", "require", ..., ..., ..., "__bbox", "result", "data"),
+            expected_type=dict,
+        )
+        or []
+    )
+    media = traverse_obj(
+        post,
+        (
+            ...,
+            "attachments",
+            ...,
+            lambda k, v: (k == "media" and str(v["id"]) == video_id and v["__typename"] == "Video"),
+        ),
+        expected_type=dict,
+    )
+    title = get_first(media, ("title", "text"))
+    description = get_first(media, ("creation_story", "comet_sections", "message", "story", "message", "text"))
+    page_title = title or self._html_search_regex(
+        (
+            r'<h2\s+[^>]*class="uiHeaderTitle"[^>]*>(?P<content>[^<]*)</h2>',
+            r'(?s)<span class="fbPhotosPhotoCaption".*?id="fbPhotoPageCaption"><span class="hasCaption">(?P<content>.*?)</span>',
+            self._meta_regex("og:title"),
+            self._meta_regex("twitter:title"),
+            r"<title>(?P<content>.+?)</title>",
+        ),
+        webpage,
+        "title",
+        default=None,
+        group="content",
+    )
    description = description or self._html_search_meta(
-        ['description', 'og:description', 'twitter:description'],
-        webpage, 'description', default=None)
+        ["description", "og:description", "twitter:description"], webpage, "description", default=None
+    )
    uploader_data = (
-        get_first(media, ('owner', {dict}))
-        or get_first(post, ('video', 'creation_story', 'attachments', ..., 'media', lambda k, v: k == 'owner' and v['name']))
-        or get_first(post, (..., 'video', lambda k, v: k == 'owner' and v['name']))
-        or get_first(post, ('node', 'actors', ..., {dict}))
-        or get_first(post, ('event', 'event_creator', {dict}))
-        or get_first(post, ('video', 'creation_story', 'short_form_video_context', 'video_owner', {dict})) or {})
-    uploader = uploader_data.get('name') or (
-        clean_html(get_element_by_id('fbPhotoPageAuthorName', webpage))
+        get_first(media, ("owner", {dict}))
+        or get_first(
+            post, ("video", "creation_story", "attachments", ..., "media", lambda k, v: k == "owner" and v["name"])
+        )
+        or get_first(post, (..., "video", lambda k, v: k == "owner" and v["name"]))
+        or get_first(post, ("node", "actors", ..., {dict}))
+        or get_first(post, ("event", "event_creator", {dict}))
+        or get_first(post, ("video", "creation_story", "short_form_video_context", "video_owner", {dict}))
+        or {}
+    )
+    uploader = uploader_data.get("name") or (
+        clean_html(get_element_by_id("fbPhotoPageAuthorName", webpage))
        or self._search_regex(
-            (r'ownerName\s*:\s*"([^"]+)"', *self._og_regexes('title')), webpage, 'uploader', fatal=False))
-    timestamp = int_or_none(self._search_regex(
-        r'<abbr[^>]+data-utime=["\'](\d+)', webpage,
-        'timestamp', default=None))
-    thumbnail = self._html_search_meta(
-        ['og:image', 'twitter:image'], webpage, 'thumbnail', default=None)
+            (r'ownerName\s*:\s*"([^"]+)"', *self._og_regexes("title")), webpage, "uploader", fatal=False
+        )
+    )
+    timestamp = int_or_none(self._search_regex(r'<abbr[^>]+data-utime=["\'](\d+)', webpage, "timestamp", default=None))
+    thumbnail = self._html_search_meta(["og:image", "twitter:image"], webpage, "thumbnail", default=None)
    # some webpages contain unretrievable thumbnail urls
    # like https://lookaside.fbsbx.com/lookaside/crawler/media/?media_id=10155168902769113&get_thumbnail=1
    # in https://www.facebook.com/yaroslav.korpan/videos/1417995061575415/
-    if thumbnail and not re.search(r'\.(?:jpg|png)', thumbnail):
+    if thumbnail and not re.search(r"\.(?:jpg|png)", thumbnail):
        thumbnail = None
    info_dict = {
-        'description': description,
-        'uploader': uploader,
-        'uploader_id': uploader_data.get('id'),
-        'timestamp': timestamp,
-        'thumbnail': thumbnail,
-        'view_count': parse_count(self._search_regex(
-            (r'\bviewCount\s*:\s*["\']([\d,.]+)', r'video_view_count["\']\s*:\s*(\d+)'),
-            webpage, 'view count', default=None)),
-        'concurrent_view_count': get_first(post, (
-            ('video', (..., ..., 'attachments', ..., 'media')), 'liveViewerCount', {int_or_none})),
-        **traverse_obj(post, (lambda _, v: video_id in v['url'], 'feedback', {
-            'like_count': ('likers', 'count', {int}),
-            'comment_count': ('total_comment_count', {int}),
-            'repost_count': ('share_count_reduced', {parse_count}),
-        }), get_all=False),
+        "description": description,
+        "uploader": uploader,
+        "uploader_id": uploader_data.get("id"),
+        "timestamp": timestamp,
+        "thumbnail": thumbnail,
+        "view_count": parse_count(
+            self._search_regex(
+                (r'\bviewCount\s*:\s*["\']([\d,.]+)', r'video_view_count["\']\s*:\s*(\d+)'),
+                webpage,
+                "view count",
+                default=None,
+            )
+        ),
+        "concurrent_view_count": get_first(
+            post, (("video", (..., ..., "attachments", ..., "media")), "liveViewerCount", {int_or_none})
+        ),
+        **traverse_obj(
+            post,
+            (
+                lambda _, v: video_id in v["url"],
+                "feedback",
+                {
+                    "like_count": ("likers", "count", {int}),
+                    "comment_count": ("total_comment_count", {int}),
+                    "repost_count": ("share_count_reduced", {parse_count}),
+                },
+            ),
+            get_all=False,
+        ),
    }

    info_json_ld = self._search_json_ld(webpage, video_id, default={})
-    info_json_ld['title'] = (re.sub(r'\s*\|\s*Facebook$', '', title or info_json_ld.get('title') or page_title or '')
-                                or (description or '').replace('\n', ' ') or f'Facebook video #{video_id}')
+    info_json_ld["title"] = (
+        re.sub(r"\s*\|\s*Facebook$", "", title or info_json_ld.get("title") or page_title or "")
+        or (description or "").replace("\n", " ")
+        or f"Facebook video #{video_id}"
+    )
    return merge_dicts(info_json_ld, info_dict)
-class Facebook(GenericDropin):
-    
-    def extract_post(self, url: str, ie_instance):

-        post_id_regex = r'(?P<id>pfbid[A-Za-z0-9]+|\d+|t\.(\d+\/\d+))'
-        post_id = re.search(post_id_regex, url).group('id')
-        webpage = ie_instance._download_webpage(
-            url.replace('://m.facebook.com/', '://www.facebook.com/'), post_id)
+
+class Facebook(GenericDropin):
+    def extract_post(self, url: str, ie_instance):
+        video_id = ie_instance._match_valid_url(url).group("id")
+        ie_instance._download_webpage(url.replace("://m.facebook.com/", "://www.facebook.com/"), video_id)
+        webpage = ie_instance._download_webpage(url, ie_instance._match_valid_url(url).group("id"))
+
+        post_id_regex = r"(?P<id>pfbid[A-Za-z0-9]+|\d+|t\.(\d+\/\d+))"
+        post_id = re.search(post_id_regex, url).group("id")
+        webpage = ie_instance._download_webpage(url.replace("://m.facebook.com/", "://www.facebook.com/"), post_id)

        # TODO: For long posts, this _extract_metadata only seems to return the first 100 or so characters, followed by ...

@@ -93,20 +139,19 @@ class Facebook(GenericDropin):

    def create_metadata(self, post: dict, ie_instance, archiver, url):
        result = Metadata()
-        result.set_content(post.get('description', ''))
-        result.set_title(post.get('title', ''))
-        result.set('author', post.get('uploader', ''))
+        result.set_content(post.get("description", ""))
+        result.set_title(post.get("title", ""))
+        result.set("author", post.get("uploader", ""))
        result.set_url(url)
        return result
-    
+
    def is_suitable(self, url, info_extractor):
-        regex = r'(?:https?://(?:[\w-]+\.)?(?:facebook\.com||facebookwkhpilnemxj7asaniu7vnjjbiltxjqhye3mhbshg7kx5tfyd\.onion)/)'
+        regex = r"(?:https?://(?:[\w-]+\.)?(?:facebook\.com||facebookwkhpilnemxj7asaniu7vnjjbiltxjqhye3mhbshg7kx5tfyd\.onion)/)"
        return re.match(regex, url)
-    
+
    def skip_ytdlp_download(self, url: str, ie_instance):
        """
        Skip using the ytdlp download method for Facebook *photo* posts, they have a URL with an id of t.XXXXX/XXXXX
        """
-        if re.search(r'/t.\d+/\d+', url):
+        if re.search(r"/t.\d+/\d+", url):
            return True
-
--- a/src/auto_archiver/modules/generic_extractor/generic_extractor.py
+++ b/src/auto_archiver/modules/generic_extractor/generic_extractor.py
@@ -1,18 +1,68 @@
-import datetime, os, yt_dlp, pysubs2
+import datetime
+import os
 import importlib
+import subprocess
+
 from typing import Generator, Type
+
+import yt_dlp
 from yt_dlp.extractor.common import InfoExtractor
+import pysubs2

 from loguru import logger

 from auto_archiver.core.extractor import Extractor
 from auto_archiver.core import Metadata, Media

-class Skip(Exception):
+
+class SkipYtdlp(Exception):
    pass
+
+
 class GenericExtractor(Extractor):
    _dropins = {}

+    def setup(self):
+        # check for file .ytdlp-update in the secrets folder
+        if self.ytdlp_update_interval < 0:
+            return
+
+        use_secrets = os.path.exists("secrets")
+        path = os.path.join("secrets" if use_secrets else "", ".ytdlp-update")
+        next_update_check = None
+        if os.path.exists(path):
+            with open(path, "r") as f:
+                next_update_check = datetime.datetime.fromisoformat(f.read())
+
+        if not next_update_check or next_update_check < datetime.datetime.now():
+            self.update_ytdlp()
+
+            next_update_check = datetime.datetime.now() + datetime.timedelta(days=self.ytdlp_update_interval)
+            with open(path, "w") as f:
+                f.write(next_update_check.isoformat())
+
+    def update_ytdlp(self):
+        logger.info("Checking and updating yt-dlp...")
+        logger.info(
+            f"Tip: change the 'ytdlp_update_interval' setting to control how often yt-dlp is updated. Set to -1 to disable or 0 to enable on every run. Current setting: {self.ytdlp_update_interval}"
+        )
+        from importlib.metadata import version as get_version
+
+        old_version = get_version("yt-dlp")
+        try:
+            # try and update with pip (this works inside poetry environment and in a normal virtualenv)
+            result = subprocess.run(["pip", "install", "--upgrade", "yt-dlp"], check=True, capture_output=True)
+
+            if "Successfully installed yt-dlp" in result.stdout.decode():
+                new_version = importlib.metadata.version("yt-dlp")
+                logger.info(f"yt-dlp successfully (from {old_version} to {new_version})")
+                importlib.reload(yt_dlp)
+            else:
+                logger.info("yt-dlp already up to date")
+
+        except Exception as e:
+            logger.error(f"Error updating yt-dlp: {e}")
+
    def suitable_extractors(self, url: str) -> Generator[str, None, None]:
        """
        Returns a list of valid extractors for the given URL"""
@@ -29,17 +79,17 @@ class GenericExtractor(Extractor):
            if info_extractor.suitable(url):
                yield info_extractor
                continue
-            

-        
    def suitable(self, url: str) -> bool:
        """
        Checks for valid URLs out of all ytdlp extractors.
        Returns False for the GenericIE, which as labelled by yt-dlp: 'Generic downloader that works on some sites'
        """
        return any(self.suitable_extractors(url))
-    
-    def download_additional_media(self, video_data: dict, info_extractor: InfoExtractor, metadata: Metadata) -> Metadata:
+
+    def download_additional_media(
+        self, video_data: dict, info_extractor: InfoExtractor, metadata: Metadata
+    ) -> Metadata:
        """
        Downloads additional media like images, comments, subtitles, etc.

@@ -48,7 +98,7 @@ class GenericExtractor(Extractor):

        # Just get the main thumbnail. More thumbnails are available in
        # video_data['thumbnails'] should they be required
-        thumbnail_url = video_data.get('thumbnail')
+        thumbnail_url = video_data.get("thumbnail")
        if thumbnail_url:
            try:
                cover_image_path = self.download_from_url(thumbnail_url)
@@ -71,15 +121,65 @@ class GenericExtractor(Extractor):
        Clean up the ytdlp generic video data to make it more readable and remove unnecessary keys that ytdlp adds
        """

-        base_keys = ['formats', 'thumbnail', 'display_id', 'epoch', 'requested_downloads',
-                     'duration_string', 'thumbnails', 'http_headers', 'webpage_url_basename', 'webpage_url_domain',
-                     'extractor', 'extractor_key', 'playlist', 'playlist_index', 'duration_string', 'protocol', 'requested_subtitles',
-                     'format_id', 'acodec', 'vcodec', 'ext', 'epoch', '_has_drm', 'filesize', 'audio_ext', 'video_ext', 'vbr', 'abr',
-                     'resolution', 'dynamic_range', 'aspect_ratio', 'cookies', 'format', 'quality', 'preference', 'artists',
-                     'channel_id', 'subtitles', 'tbr', 'url', 'original_url', 'automatic_captions', 'playable_in_embed', 'live_status',
-                     '_format_sort_fields', 'chapters', 'requested_formats', 'format_note',
-                     'audio_channels', 'asr', 'fps', 'was_live', 'is_live', 'heatmap', 'age_limit', 'stretched_ratio']
-        
+        base_keys = [
+            "formats",
+            "thumbnail",
+            "display_id",
+            "epoch",
+            "requested_downloads",
+            "duration_string",
+            "thumbnails",
+            "http_headers",
+            "webpage_url_basename",
+            "webpage_url_domain",
+            "extractor",
+            "extractor_key",
+            "playlist",
+            "playlist_index",
+            "duration_string",
+            "protocol",
+            "requested_subtitles",
+            "format_id",
+            "acodec",
+            "vcodec",
+            "ext",
+            "epoch",
+            "_has_drm",
+            "filesize",
+            "audio_ext",
+            "video_ext",
+            "vbr",
+            "abr",
+            "resolution",
+            "dynamic_range",
+            "aspect_ratio",
+            "cookies",
+            "format",
+            "quality",
+            "preference",
+            "artists",
+            "channel_id",
+            "subtitles",
+            "tbr",
+            "url",
+            "original_url",
+            "automatic_captions",
+            "playable_in_embed",
+            "live_status",
+            "_format_sort_fields",
+            "chapters",
+            "requested_formats",
+            "format_note",
+            "audio_channels",
+            "asr",
+            "fps",
+            "was_live",
+            "is_live",
+            "heatmap",
+            "age_limit",
+            "stretched_ratio",
+        ]
+
        dropin = self.dropin_for_name(info_extractor.ie_key())
        if dropin:
            try:
@@ -88,8 +188,8 @@ class GenericExtractor(Extractor):
                pass

        return base_keys
-    
-    def add_metadata(self, video_data: dict, info_extractor: InfoExtractor, url:str, result: Metadata) -> Metadata:
+
+    def add_metadata(self, video_data: dict, info_extractor: InfoExtractor, url: str, result: Metadata) -> Metadata:
        """
        Creates a Metadata object from the given video_data
        """
@@ -98,29 +198,36 @@ class GenericExtractor(Extractor):
        result = self.download_additional_media(video_data, info_extractor, result)

        # keep both 'title' and 'fulltitle', but prefer 'title', falling back to 'fulltitle' if it doesn't exist
-        result.set_title(video_data.pop('title', video_data.pop('fulltitle', "")))
+        result.set_title(video_data.pop("title", video_data.pop("fulltitle", "")))
        result.set_url(url)
-
+        if "description" in video_data:
+            result.set_content(video_data["description"])
        # extract comments if enabled
        if self.comments:
-            result.set("comments", [{
-                "text": c["text"],
-                "author": c["author"], 
-                "timestamp": datetime.datetime.fromtimestamp(c.get("timestamp"), tz = datetime.timezone.utc)
-            } for c in video_data.get("comments", [])])
+            result.set(
+                "comments",
+                [
+                    {
+                        "text": c["text"],
+                        "author": c["author"],
+                        "timestamp": datetime.datetime.fromtimestamp(c.get("timestamp"), tz=datetime.timezone.utc),
+                    }
+                    for c in video_data.get("comments", [])
+                ],
+            )

        # then add the common metadata
        if timestamp := video_data.pop("timestamp", None):
-            timestamp = datetime.datetime.fromtimestamp(timestamp, tz = datetime.timezone.utc).isoformat()
+            timestamp = datetime.datetime.fromtimestamp(timestamp, tz=datetime.timezone.utc).isoformat()
            result.set_timestamp(timestamp)
        if upload_date := video_data.pop("upload_date", None):
-            upload_date = datetime.datetime.strptime(upload_date, '%Y%m%d').replace(tzinfo=datetime.timezone.utc)
+            upload_date = datetime.datetime.strptime(upload_date, "%Y%m%d").replace(tzinfo=datetime.timezone.utc)
            result.set("upload_date", upload_date)
-        
+
        # then clean away any keys we don't want
        for clean_key in self.keys_to_clean(info_extractor, video_data):
            video_data.pop(clean_key, None)
-        
+
        # then add the rest of the video data
        for k, v in video_data.items():
            if v:
@@ -138,26 +245,28 @@ class GenericExtractor(Extractor):

        if not dropin:
            # TODO: add a proper link to 'how to create your own dropin'
-            logger.debug(f"""Could not find valid dropin for {info_extractor.IE_NAME}.
+            logger.debug(f"""Could not find valid dropin for {info_extractor.ie_key()}.
                     Why not try creating your own, and make sure it has a valid function called 'create_metadata'. Learn more: https://auto-archiver.readthedocs.io/en/latest/user_guidelines.html#""")
            return False
-        
+
        post_data = dropin.extract_post(url, ie_instance)
        result = dropin.create_metadata(post_data, ie_instance, self, url)
        return self.add_metadata(post_data, info_extractor, url, result)

-    def get_metadata_for_video(self, data: dict, info_extractor: Type[InfoExtractor], url: str, ydl: yt_dlp.YoutubeDL) -> Metadata:
-
+    def get_metadata_for_video(
+        self, data: dict, info_extractor: Type[InfoExtractor], url: str, ydl: yt_dlp.YoutubeDL
+    ) -> Metadata:
        # this time download
-        ydl.params['getcomments'] = self.comments
-        #TODO: for playlist or long lists of videos, how to download one at a time so they can be stored before the next one is downloaded?
+        ydl.params["getcomments"] = self.comments
+        # TODO: for playlist or long lists of videos, how to download one at a time so they can be stored before the next one is downloaded?
        data = ydl.extract_info(url, ie_key=info_extractor.ie_key(), download=True)
        if "entries" in data:
            entries = data.get("entries", [])
            if not len(entries):
-                logger.warning('YoutubeDLArchiver could not find any video')
+                logger.warning("YoutubeDLArchiver could not find any video")
                return False
-        else: entries = [data]
+        else:
+            entries = [data]

        result = Metadata()

@@ -165,17 +274,18 @@ class GenericExtractor(Extractor):
            try:
                filename = ydl.prepare_filename(entry)
                if not os.path.exists(filename):
-                    filename = filename.split('.')[0] + '.mkv'
+                    filename = filename.split(".")[0] + ".mkv"

                new_media = Media(filename)
                for x in ["duration", "original_url", "fulltitle", "description", "upload_date"]:
-                    if x in entry: new_media.set(x, entry[x])
+                    if x in entry:
+                        new_media.set(x, entry[x])

                # read text from subtitles if enabled
                if self.subtitles:
-                    for lang, val in (data.get('requested_subtitles') or {}).items():
-                        try:    
-                            subs = pysubs2.load(val.get('filepath'), encoding="utf-8")
+                    for lang, val in (data.get("requested_subtitles") or {}).items():
+                        try:
+                            subs = pysubs2.load(val.get("filepath"), encoding="utf-8")
                            text = " ".join([line.text for line in subs])
                            new_media.set(f"subtitles_{lang}", text)
                        except Exception as e:
@@ -185,8 +295,8 @@ class GenericExtractor(Extractor):
                logger.error(f"Error processing entry {entry}: {e}")

        return self.add_metadata(data, info_extractor, url, result)
-    
-    def dropin_for_name(self, dropin_name: str, additional_paths = [], package=__package__) -> Type[InfoExtractor]:
+
+    def dropin_for_name(self, dropin_name: str, additional_paths=[], package=__package__) -> Type[InfoExtractor]:
        dropin_name = dropin_name.lower()

        if dropin_name == "generic":
@@ -194,6 +304,7 @@ class GenericExtractor(Extractor):
            return None

        dropin_class_name = dropin_name.title()
+
        def _load_dropin(dropin):
            dropin_class = getattr(dropin, dropin_class_name)()
            dropin.extractor = self
@@ -218,7 +329,7 @@ class GenericExtractor(Extractor):
                return _load_dropin(dropin)
            except (FileNotFoundError, ModuleNotFoundError):
                pass
-        
+
        # fallback to loading the dropins within auto-archiver
        try:
            return _load_dropin(importlib.import_module(f".{dropin_name}", package=package))
@@ -230,46 +341,53 @@ class GenericExtractor(Extractor):
    def download_for_extractor(self, info_extractor: InfoExtractor, url: str, ydl: yt_dlp.YoutubeDL) -> Metadata:
        """
        Tries to download the given url using the specified extractor
-        
+
        It first tries to use ytdlp directly to download the video. If the post is not a video, it will then try to
        use the extractor's _extract_post method to get the post metadata if possible.
        """
        # when getting info without download, we also don't need the comments
-        ydl.params['getcomments'] = False
+        ydl.params["getcomments"] = False
        result = False

        dropin_submodule = self.dropin_for_name(info_extractor.ie_key())

        try:
-            if dropin_submodule and dropin_submodule.skip_ytdlp_download(url, info_extractor):
-                logger.debug(f"Skipping using ytdlp to download files for {info_extractor.ie_key()} (dropin override)")
-                raise Skip()
+            if dropin_submodule and dropin_submodule.skip_ytdlp_download(info_extractor, url):
+                logger.debug(f"Skipping using ytdlp to download files for {info_extractor.ie_key()}")
+                raise SkipYtdlp()

            # don't download since it can be a live stream
            data = ydl.extract_info(url, ie_key=info_extractor.ie_key(), download=False)
-            if data.get('is_live', False) and not self.livestreams:
+            if data.get("is_live", False) and not self.livestreams:
                logger.warning("Livestream detected, skipping due to 'livestreams' configuration setting")
                return False
            # it's a valid video, that the youtubdedl can download out of the box
            result = self.get_metadata_for_video(data, info_extractor, url, ydl)

        except Exception as e:
-            if info_extractor.ie_key() == "generic":
+            if info_extractor.IE_NAME == "generic":
                # don't clutter the logs with issues about the 'generic' extractor not having a dropin
                return False
-            
-            if not isinstance(e, Skip):
-                logger.debug(f'Issue using "{info_extractor.IE_NAME}" extractor to download video (error: {repr(e)}), attempting to use dropin to get post data instead')
+
+            if not isinstance(e, SkipYtdlp):
+                logger.debug(
+                    f'Issue using "{info_extractor.IE_NAME}" extractor to download video (error: {repr(e)}), attempting to use dropin to get post data instead'
+                )

            try:
                result = self.get_metadata_for_post(info_extractor, url, ydl)
            except (yt_dlp.utils.DownloadError, yt_dlp.utils.ExtractorError) as post_e:
-                logger.error(f'Error downloading metadata for post: {post_e}')
+                logger.error("Error downloading metadata for post: {error}", error=str(post_e))
                return False
            except Exception as generic_e:
-                logger.debug(f'Attempt to extract using ytdlp dropin for "{info_extractor.IE_NAME}" failed:  \n  {repr(generic_e)}', exc_info=True)
+                logger.debug(
+                    'Attempt to extract using ytdlp extractor "{name}" failed:  \n  {error}',
+                    name=info_extractor.IE_NAME,
+                    error=str(generic_e),
+                    exc_info=True,
+                )
                return False
-        
+
        if result:
            extractor_name = "yt-dlp"
            if info_extractor:
@@ -285,42 +403,56 @@ class GenericExtractor(Extractor):
    def download(self, item: Metadata) -> Metadata:
        url = item.get_url()

-        #TODO: this is a temporary hack until this issue is closed: https://github.com/yt-dlp/yt-dlp/issues/11025
+        # TODO: this is a temporary hack until this issue is closed: https://github.com/yt-dlp/yt-dlp/issues/11025
        if url.startswith("https://ya.ru"):
            url = url.replace("https://ya.ru", "https://yandex.ru")
            item.set("replaced_url", url)

+        ydl_options = [
+            "-o",
+            os.path.join(self.tmp_dir, "%(id)s.%(ext)s"),
+            "--quiet",
+            "--no-playlist" if not self.allow_playlist else "--yes-playlist",
+            "--write-subs" if self.subtitles else "--no-write-subs",
+            "--write-auto-subs" if self.subtitles else "--no-write-auto-subs",
+            "--live-from-start" if self.live_from_start else "--no-live-from-start",
+            "--proxy",
+            self.proxy if self.proxy else "",
+            f"--max-downloads {self.max_downloads}" if self.max_downloads != "inf" else "",
+            f"--playlist-end {self.max_downloads}" if self.max_downloads != "inf" else "",
+        ]

-        ydl_options = {'outtmpl': os.path.join(self.tmp_dir, f'%(id)s.%(ext)s'), 
-                       'quiet': False, 'noplaylist': not self.allow_playlist ,
-                       'writesubtitles': self.subtitles,'writeautomaticsub': self.subtitles,
-                       "live_from_start": self.live_from_start, "proxy": self.proxy,
-                       "max_downloads": self.max_downloads, "playlistend": self.max_downloads}
-        
        # set up auth
        auth = self.auth_for_site(url, extract_cookies=False)
+
        # order of importance: username/pasword -> api_key -> cookie -> cookies_from_browser -> cookies_file
        if auth:
-            if 'username' in auth and 'password' in auth:
-                logger.debug(f'Using provided auth username and password for {url}')
-                ydl_options['username'] = auth['username']
-                ydl_options['password'] = auth['password']
-            elif 'cookie' in auth:
-                logger.debug(f'Using provided auth cookie for {url}')
-                yt_dlp.utils.std_headers['cookie'] = auth['cookie']
-            elif 'cookies_from_browser' in auth:
-                logger.debug(f'Using extracted cookies from browser {self.cookies_from_browser} for {url}')
-                ydl_options['cookiesfrombrowser'] = auth['cookies_from_browser']
-            elif 'cookies_file' in auth:
-                logger.debug(f'Using cookies from file {self.cookie_file} for {url}')
-                ydl_options['cookiesfile'] = auth['cookies_file']
+            if "username" in auth and "password" in auth:
+                logger.debug(f"Using provided auth username and password for {url}")
+                ydl_options.extend(("--username", auth["username"]))
+                ydl_options.extend(("--password", auth["password"]))
+            elif "cookie" in auth:
+                logger.debug(f"Using provided auth cookie for {url}")
+                yt_dlp.utils.std_headers["cookie"] = auth["cookie"]
+            elif "cookies_from_browser" in auth:
+                logger.debug(f"Using extracted cookies from browser {auth['cookies_from_browser']} for {url}")
+                ydl_options.extend(("--cookies-from-browser", auth["cookies_from_browser"]))
+            elif "cookies_file" in auth:
+                logger.debug(f"Using cookies from file {auth['cookies_file']} for {url}")
+                ydl_options.extend(("--cookies", auth["cookies_file"]))

-        ydl = yt_dlp.YoutubeDL(ydl_options) # allsubtitles and subtitleslangs not working as expected, so default lang is always "en"
+        if self.ytdlp_args:
+            logger.debug("Adding additional ytdlp arguments: {self.ytdlp_args}")
+            ydl_options += self.ytdlp_args.split(" ")
+
+        *_, validated_options = yt_dlp.parse_options(ydl_options)
+        ydl = yt_dlp.YoutubeDL(
+            validated_options
+        )  # allsubtitles and subtitleslangs not working as expected, so default lang is always "en"

        for info_extractor in self.suitable_extractors(url):
            result = self.download_for_extractor(info_extractor, url, ydl)
            if result:
                return result

-
        return False
--- a/src/auto_archiver/modules/generic_extractor/tiktok.py
+++ b/src/auto_archiver/modules/generic_extractor/tiktok.py
@@ -0,0 +1,72 @@
+import requests
+from loguru import logger
+from auto_archiver.core import Metadata, Media
+from datetime import datetime, timezone
+from .dropin import GenericDropin
+
+
+class Tiktok(GenericDropin):
+    """
+    TikTok droping for the Generic Extractor that uses an unofficial API if/when ytdlp fails.
+    It's useful for capturing content that requires a login, like sensitive content.
+    """
+
+    TIKWM_ENDPOINT = "https://www.tikwm.com/api/?url={url}"
+
+    def extract_post(self, url: str, ie_instance):
+        logger.debug(f"Using Tikwm API to attempt to download tiktok video from {url=}")
+
+        endpoint = self.TIKWM_ENDPOINT.format(url=url)
+
+        r = requests.get(endpoint)
+        if r.status_code != 200:
+            raise ValueError(f"unexpected status code '{r.status_code}' from tikwm.com for {url=}:")
+
+        try:
+            json_response = r.json()
+        except ValueError:
+            raise ValueError(f"failed to parse JSON response from tikwm.com for {url=}")
+
+        if not json_response.get("msg") == "success" or not (api_data := json_response.get("data", {})):
+            raise ValueError(f"failed to get a valid response from tikwm.com for {url=}: {repr(json_response)}")
+
+        # tries to get the non-watermarked version first
+        video_url = api_data.pop("play", api_data.pop("wmplay", None))
+        if not video_url:
+            raise ValueError(f"no valid video URL found in response from tikwm.com for {url=}")
+
+        api_data["video_url"] = video_url
+        return api_data
+
+    def create_metadata(self, post: dict, ie_instance, archiver, url):
+        # prepare result, start by downloading video
+        result = Metadata()
+        video_url = post.pop("video_url")
+
+        # get the cover if possible
+        cover_url = post.pop("origin_cover", post.pop("cover", post.pop("ai_dynamic_cover", None)))
+        if cover_url and (cover_downloaded := archiver.download_from_url(cover_url)):
+            result.add_media(Media(cover_downloaded))
+
+        # get the video or fail
+        video_downloaded = archiver.download_from_url(video_url, f"vid_{post.get('id', '')}")
+        if not video_downloaded:
+            logger.error(f"failed to download video from {video_url}")
+            return False
+        video_media = Media(video_downloaded)
+        if duration := post.pop("duration", None):
+            video_media.set("duration", duration)
+        result.add_media(video_media)
+
+        # add remaining metadata
+        result.set_title(post.pop("title", ""))
+
+        if created_at := post.pop("create_time", None):
+            result.set_timestamp(datetime.fromtimestamp(created_at, tz=timezone.utc))
+
+        if author := post.pop("author", None):
+            result.set("author", author)
+
+        result.set("api_data", post)
+
+        return result
--- a/src/auto_archiver/modules/generic_extractor/truth.py
+++ b/src/auto_archiver/modules/generic_extractor/truth.py
@@ -9,11 +9,11 @@ from dateutil.parser import parse as parse_dt

 from .dropin import GenericDropin

-class Truth(GenericDropin):

+class Truth(GenericDropin):
    def extract_post(self, url, ie_instance: InfoExtractor) -> dict:
        video_id = ie_instance._match_id(url)
-        truthsocial_url = f'https://truthsocial.com/api/v1/statuses/{video_id}'
+        truthsocial_url = f"https://truthsocial.com/api/v1/statuses/{video_id}"
        return ie_instance._download_json(truthsocial_url, video_id)

    def skip_ytdlp_download(self, url, ie_instance: Type[InfoExtractor]) -> bool:
@@ -22,31 +22,42 @@ class Truth(GenericDropin):
    def create_metadata(self, post: dict, ie_instance: InfoExtractor, archiver: Extractor, url: str) -> Metadata:
        """
        Creates metadata from a truth social post
-        
+
        Only used for posts that contain no media. ytdlp.TruthIE extractor can handle posts with media
-        
+
        Format is:
-        
+
        {'id': '109598702184774628', 'created_at': '2022-12-29T19:51:18.161Z', 'in_reply_to_id': None, 'quote_id': None, 'in_reply_to_account_id': None, 'sensitive': False, 'spoiler_text': '', 'visibility': 'public', 'language': 'en', 'uri': 'https://truthsocial.com/@bbcnewa/109598702184774628', 'url': 'https://truthsocial.com/@bbcnewa/109598702184774628', 'content': '<p>Pele, regarded by many as football\'s greatest ever player, has died in Brazil at the age of 82. <a href="https://www.bbc.com/sport/football/42751517" rel="nofollow noopener noreferrer" target="_blank"><span class="invisible">https://www.</span><span class="ellipsis">bbc.com/sport/football/4275151</span><span class="invisible">7</span></a></p>', 'account': {'id': '107905163010312793', 'username': 'bbcnewa', 'acct': 'bbcnewa', 'display_name': 'BBC News', 'locked': False, 'bot': False, 'discoverable': True, 'group': False, 'created_at': '2022-03-05T17:42:01.159Z', 'note': '<p>News, features and analysis by the BBC</p>', 'url': 'https://truthsocial.com/@bbcnewa', 'avatar': 'https://static-assets-1.truthsocial.com/tmtg:prime-ts-assets/accounts/avatars/107/905/163/010/312/793/original/e7c07550dc22c23a.jpeg', 'avatar_static': 'https://static-assets-1.truthsocial.com/tmtg:prime-ts-assets/accounts/avatars/107/905/163/010/312/793/original/e7c07550dc22c23a.jpeg', 'header': 'https://static-assets-1.truthsocial.com/tmtg:prime-ts-assets/accounts/headers/107/905/163/010/312/793/original/a00eeec2b57206c7.jpeg', 'header_static': 'https://static-assets-1.truthsocial.com/tmtg:prime-ts-assets/accounts/headers/107/905/163/010/312/793/original/a00eeec2b57206c7.jpeg', 'followers_count': 1131, 'following_count': 3, 'statuses_count': 9, 'last_status_at': '2024-11-12', 'verified': False, 'location': '', 'website': 'https://www.bbc.com/news', 'unauth_visibility': True, 'chats_onboarded': True, 'feeds_onboarded': True, 'accepting_messages': False, 'show_nonmember_group_statuses': None, 'emojis': [], 'fields': [], 'tv_onboarded': True, 'tv_account': False}, 'media_attachments': [], 'mentions': [], 'tags': [], 'card': None, 'group': None, 'quote': None, 'in_reply_to': None, 'reblog': None, 'sponsored': False, 'replies_count': 1, 'reblogs_count': 0, 'favourites_count': 2, 'favourited': False, 'reblogged': False, 'muted': False, 'pinned': False, 'bookmarked': False, 'poll': None, 'emojis': []}
        """

        result = Metadata()
        result.set_url(url)
-        timestamp = post['created_at'] # format is 2022-12-29T19:51:18.161Z
+        timestamp = post["created_at"]  # format is 2022-12-29T19:51:18.161Z
        result.set_timestamp(parse_dt(timestamp))
-        result.set('description', post['content'])
-        result.set('author', post['account']['username'])
+        result.set("description", post["content"])
+        result.set("author", post["account"]["username"])

-        for key in ['replies_count', 'reblogs_count', 'favourites_count', ('account', 'followers_count'), ('account', 'following_count'), ('account', 'statuses_count'), ('account', 'display_name'), 'language', 'in_reply_to_account', 'replies_count']:
+        for key in [
+            "replies_count",
+            "reblogs_count",
+            "favourites_count",
+            ("account", "followers_count"),
+            ("account", "following_count"),
+            ("account", "statuses_count"),
+            ("account", "display_name"),
+            "language",
+            "in_reply_to_account",
+            "replies_count",
+        ]:
            if isinstance(key, tuple):
                store_key = " ".join(key)
            else:
                store_key = key
            result.set(store_key, traverse_obj(post, key))
-        
-        # add the media
-        for media in post.get('media_attachments', []):
-            filename = archiver.download_from_url(media['url'])
-            result.add_media(Media(filename), id=media.get('id'))

-        return result
+        # add the media
+        for media in post.get("media_attachments", []):
+            filename = archiver.download_from_url(media["url"])
+            result.add_media(Media(filename), id=media.get("id"))
+
+        return result
--- a/src/auto_archiver/modules/generic_extractor/twitter.py
+++ b/src/auto_archiver/modules/generic_extractor/twitter.py
@@ -1,4 +1,6 @@
-import re, mimetypes, json
+import re
+import mimetypes
+import json
 from datetime import datetime

 from loguru import logger
@@ -10,9 +12,8 @@ from auto_archiver.core.extractor import Extractor

 from .dropin import GenericDropin, InfoExtractor

+
 class Twitter(GenericDropin):
-
-
    def choose_variant(self, variants):
        # choosing the highest quality possible
        variant, width, height = None, 0, 0
@@ -27,44 +28,43 @@ class Twitter(GenericDropin):
            else:
                variant = var if not variant else variant
        return variant
-    
+
    def extract_post(self, url: str, ie_instance: InfoExtractor):
-        twid = ie_instance._match_valid_url(url).group('id')
+        twid = ie_instance._match_valid_url(url).group("id")
        return ie_instance._extract_status(twid=twid)

    def create_metadata(self, tweet: dict, ie_instance: InfoExtractor, archiver: Extractor, url: str) -> Metadata:
        result = Metadata()
        try:
            if not tweet.get("user") or not tweet.get("created_at"):
-                raise ValueError(f"Error retreiving post. Are you sure it exists?")
+                raise ValueError("Error retreiving post. Are you sure it exists?")
            timestamp = datetime.strptime(tweet["created_at"], "%a %b %d %H:%M:%S %z %Y")
        except (ValueError, KeyError) as ex:
            logger.warning(f"Unable to parse tweet: {str(ex)}\nRetreived tweet data: {tweet}")
            return False
-                
-        result\
-            .set_title(tweet.get('full_text', ''))\
-            .set_content(json.dumps(tweet, ensure_ascii=False))\
-            .set_timestamp(timestamp)
+
+        result.set_title(tweet.get("full_text", "")).set_content(json.dumps(tweet, ensure_ascii=False)).set_timestamp(
+            timestamp
+        )
        if not tweet.get("entities", {}).get("media"):
-            logger.debug('No media found, archiving tweet text only')
+            logger.debug("No media found, archiving tweet text only")
            result.status = "twitter-ytdl"
            return result
        for i, tw_media in enumerate(tweet["entities"]["media"]):
            media = Media(filename="")
            mimetype = ""
            if tw_media["type"] == "photo":
-                media.set("src", UrlUtil.twitter_best_quality_url(tw_media['media_url_https']))
+                media.set("src", UrlUtil.twitter_best_quality_url(tw_media["media_url_https"]))
                mimetype = "image/jpeg"
            elif tw_media["type"] == "video":
-                variant = self.choose_variant(tw_media['video_info']['variants'])
-                media.set("src", variant['url'])
-                mimetype = variant['content_type']
+                variant = self.choose_variant(tw_media["video_info"]["variants"])
+                media.set("src", variant["url"])
+                mimetype = variant["content_type"]
            elif tw_media["type"] == "animated_gif":
-                variant = tw_media['video_info']['variants'][0]
-                media.set("src", variant['url'])
-                mimetype = variant['content_type']
+                variant = tw_media["video_info"]["variants"][0]
+                media.set("src", variant["url"])
+                mimetype = variant["content_type"]
            ext = mimetypes.guess_extension(mimetype)
-            media.filename = archiver.download_from_url(media.get("src"), f'{slugify(url)}_{i}{ext}')
+            media.filename = archiver.download_from_url(media.get("src"), f"{slugify(url)}_{i}{ext}")
            result.add_media(media)
-        return result
+        return result
--- a/src/auto_archiver/modules/gsheet_db/init.py
+++ b/src/auto_archiver/modules/gsheet_db/init.py
@@ -1 +0,0 @@
-from .gsheet_db import GsheetsDb
--- a/src/auto_archiver/modules/gsheet_db/manifest.py
+++ b/src/auto_archiver/modules/gsheet_db/manifest.py
@@ -1,38 +0,0 @@
-{
-    "name": "Google Sheets Database",
-    "type": ["database"],
-    "entry_point": "gsheet_db::GsheetsDb",
-    "requires_setup": True,
-    "dependencies": {
-        "python": ["loguru", "gspread", "slugify"],
-    },
-    "configs": {
-        "allow_worksheets": {
-            "default": set(),
-            "help": "(CSV) only worksheets whose name is included in allow are included (overrides worksheet_block), leave empty so all are allowed",
-        },
-        "block_worksheets": {
-            "default": set(),
-            "help": "(CSV) explicitly block some worksheets from being processed",
-        },
-        "use_sheet_names_in_stored_paths": {
-            "default": True,
-            "type": "bool",
-            "help": "if True the stored files path will include 'workbook_name/worksheet_name/...'",
-        }
-    },
-    "description": """
-    GsheetsDatabase:
-    Handles integration with Google Sheets for tracking archival tasks.
-
-### Features
- Updates a Google Sheet with the status of the archived URLs, including in progress, success or failure, and method used.
- Saves metadata such as title, text, timestamp, hashes, screenshots, and media URLs to designated columns.
- Formats media-specific metadata, such as thumbnails and PDQ hashes for the sheet.
- Skips redundant updates for empty or invalid data fields.
-
-### Notes
- Currently works only with metadata provided by GsheetFeeder. 
- Requires configuration of a linked Google Sheet and appropriate API credentials.
-    """
-}
--- a/src/auto_archiver/modules/gsheet_db/gsheet_db.py
+++ b/src/auto_archiver/modules/gsheet_db/gsheet_db.py
@@ -1,114 +0,0 @@
-from typing import Union, Tuple
-from urllib.parse import quote
-
-from loguru import logger
-
-from auto_archiver.core import Database
-from auto_archiver.core import Metadata, Media
-from auto_archiver.modules.gsheet_feeder import GWorksheet
-from auto_archiver.utils.misc import get_current_timestamp
-
-
-class GsheetsDb(Database):
-    """
-    NB: only works if GsheetFeeder is used.
-    could be updated in the future to support non-GsheetFeeder metadata
-    """
-
-    def started(self, item: Metadata) -> None:
-        logger.warning(f"STARTED {item}")
-        gw, row = self._retrieve_gsheet(item)
-        gw.set_cell(row, "status", "Archive in progress")
-
-    def failed(self, item: Metadata, reason: str) -> None:
-        logger.error(f"FAILED {item}")
-        self._safe_status_update(item, f"Archive failed {reason}")
-
-    def aborted(self, item: Metadata) -> None:
-        logger.warning(f"ABORTED {item}")
-        self._safe_status_update(item, "")
-
-    def fetch(self, item: Metadata) -> Union[Metadata, bool]:
-        """check if the given item has been archived already"""
-        return False
-
-    def done(self, item: Metadata, cached: bool = False) -> None:
-        """archival result ready - should be saved to DB"""
-        logger.success(f"DONE {item.get_url()}")
-        gw, row = self._retrieve_gsheet(item)
-        # self._safe_status_update(item, 'done')
-
-        cell_updates = []
-        row_values = gw.get_row(row)
-
-        def batch_if_valid(col, val, final_value=None):
-            final_value = final_value or val
-            try:
-                if val and gw.col_exists(col) and gw.get_cell(row_values, col) == "":
-                    cell_updates.append((row, col, final_value))
-            except Exception as e:
-                logger.error(f"Unable to batch {col}={final_value} due to {e}")
-
-        status_message = item.status
-        if cached:
-            status_message = f"[cached] {status_message}"
-        cell_updates.append((row, "status", status_message))
-
-        media: Media = item.get_final_media()
-        if hasattr(media, "urls"):
-            batch_if_valid("archive", "\n".join(media.urls))
-        batch_if_valid("date", True, get_current_timestamp())
-        batch_if_valid("title", item.get_title())
-        batch_if_valid("text", item.get("content", ""))
-        batch_if_valid("timestamp", item.get_timestamp())
-        if media:
-            batch_if_valid("hash", media.get("hash", "not-calculated"))
-
-        # merge all pdq hashes into a single string, if present
-        pdq_hashes = []
-        all_media = item.get_all_media()
-        for m in all_media:
-            if pdq := m.get("pdq_hash"):
-                pdq_hashes.append(pdq)
-        if len(pdq_hashes):
-            batch_if_valid("pdq_hash", ",".join(pdq_hashes))
-
-        if (screenshot := item.get_media_by_id("screenshot")) and hasattr(
-            screenshot, "urls"
-        ):
-            batch_if_valid("screenshot", "\n".join(screenshot.urls))
-
-        if thumbnail := item.get_first_image("thumbnail"):
-            if hasattr(thumbnail, "urls"):
-                batch_if_valid("thumbnail", f'=IMAGE("{thumbnail.urls[0]}")')
-
-        if browsertrix := item.get_media_by_id("browsertrix"):
-            batch_if_valid("wacz", "\n".join(browsertrix.urls))
-            batch_if_valid(
-                "replaywebpage",
-                "\n".join(
-                    [
-                        f"https://replayweb.page/?source={quote(wacz)}#view=pages&url={quote(item.get_url())}"
-                        for wacz in browsertrix.urls
-                    ]
-                ),
-            )
-
-        gw.batch_set_cell(cell_updates)
-
-    def _safe_status_update(self, item: Metadata, new_status: str) -> None:
-        try:
-            gw, row = self._retrieve_gsheet(item)
-            gw.set_cell(row, "status", new_status)
-        except Exception as e:
-            logger.debug(f"Unable to update sheet: {e}")
-
-    def _retrieve_gsheet(self, item: Metadata) -> Tuple[GWorksheet, int]:
-
-        if gsheet := item.get_context("gsheet"):
-            gw: GWorksheet = gsheet.get("worksheet")
-            row: int = gsheet.get("row")
-        elif self.sheet_id:
-            logger.error(f"Unable to retrieve Gsheet for {item.get_url()}, GsheetDB must be used alongside GsheetFeeder.")
-
-        return gw, row
--- a/src/auto_archiver/modules/gsheet_feeder/init.py
+++ b/src/auto_archiver/modules/gsheet_feeder/init.py
@@ -1,2 +0,0 @@
-from .gworksheet import GWorksheet
-from .gsheet_feeder import GsheetsFeeder
--- a/src/auto_archiver/modules/gsheet_feeder/gsheet_feeder.py
+++ b/src/auto_archiver/modules/gsheet_feeder/gsheet_feeder.py
@@ -1,96 +0,0 @@
-"""
-GsheetsFeeder: A Google Sheets-based feeder for the Auto Archiver.
-
-This reads data from Google Sheets and filters rows based on user-defined rules.
-The filtered rows are processed into `Metadata` objects.
-
-### Key properties
- validates the sheet's structure and filters rows based on input configurations.
- Ensures only rows with valid URLs and unprocessed statuses are included.
-"""
-import os
-import gspread
-
-from loguru import logger
-from slugify import slugify
-
-from auto_archiver.core import Feeder
-from auto_archiver.core import Metadata
-from . import GWorksheet
-
-
-class GsheetsFeeder(Feeder):
-
-    def setup(self) -> None:
-        self.gsheets_client = gspread.service_account(filename=self.service_account)
-        # TODO mv to validators
-        assert self.sheet or self.sheet_id, (
-            "You need to define either a 'sheet' name or a 'sheet_id' in your manifest."
-        )
-
-    def open_sheet(self):
-        if self.sheet:
-            return self.gsheets_client.open(self.sheet)
-        else:  # self.sheet_id
-            return self.gsheets_client.open_by_key(self.sheet_id)
-
-    def __iter__(self) -> Metadata:
-        sh = self.open_sheet()
-        for ii, worksheet in enumerate(sh.worksheets()):
-            if not self.should_process_sheet(worksheet.title):
-                logger.debug(f"SKIPPED worksheet '{worksheet.title}' due to allow/block rules")
-                continue
-            logger.info(f'Opening worksheet {ii=}: {worksheet.title=} header={self.header}')
-            gw = GWorksheet(worksheet, header_row=self.header, columns=self.columns)
-            if len(missing_cols := self.missing_required_columns(gw)):
-                logger.warning(f"SKIPPED worksheet '{worksheet.title}' due to missing required column(s) for {missing_cols}")
-                continue
-
-            # process and yield metadata here:
-            yield from self._process_rows(gw)
-            logger.success(f'Finished worksheet {worksheet.title}')
-
-    def _process_rows(self, gw: GWorksheet):
-        for row in range(1 + self.header, gw.count_rows() + 1):
-            url = gw.get_cell(row, 'url').strip()
-            if not len(url): continue
-            original_status = gw.get_cell(row, 'status')
-            status = gw.get_cell(row, 'status', fresh=original_status in ['', None])
-            # TODO: custom status parser(?) aka should_retry_from_status
-            if status not in ['', None]: continue
-
-            # All checks done - archival process starts here
-            m = Metadata().set_url(url)
-            self._set_context(m, gw, row)
-            yield m
-
-    def _set_context(self, m: Metadata, gw: GWorksheet, row: int) -> Metadata:
-        # TODO: Check folder value not being recognised
-        m.set_context("gsheet", {"row": row, "worksheet": gw})
-
-        if gw.get_cell_or_default(row, 'folder', "") is None:
-            folder = ''
-        else:
-            folder = slugify(gw.get_cell_or_default(row, 'folder', "").strip())
-        if len(folder):
-            if self.use_sheet_names_in_stored_paths:
-                m.set_context("folder", os.path.join(folder, slugify(self.sheet), slugify(gw.wks.title)))
-            else:
-                m.set_context("folder", folder)
-
-
-    def should_process_sheet(self, sheet_name: str) -> bool:
-        if len(self.allow_worksheets) and sheet_name not in self.allow_worksheets:
-            # ALLOW rules exist AND sheet name not explicitly allowed
-            return False
-        if len(self.block_worksheets) and sheet_name in self.block_worksheets:
-            # BLOCK rules exist AND sheet name is blocked
-            return False
-        return True
-
-    def missing_required_columns(self, gw: GWorksheet) -> list:
-        missing = []
-        for required_col in ['url', 'status']:
-            if not gw.col_exists(required_col):
-                missing.append(required_col)
-        return missing
--- a/src/auto_archiver/modules/gsheet_feeder_db/init.py
+++ b/src/auto_archiver/modules/gsheet_feeder_db/init.py
@@ -0,0 +1,2 @@
+from .gworksheet import GWorksheet
+from .gsheet_feeder_db import GsheetsFeederDB
--- a/src/auto_archiver/modules/gsheet_feeder_db/manifest.py
+++ b/src/auto_archiver/modules/gsheet_feeder_db/manifest.py
@@ -1,7 +1,7 @@
 {
-    "name": "Google Sheets Feeder",
-    "type": ["feeder"],
-    "entry_point": "gsheet_feeder::GsheetsFeeder",
+    "name": "Google Sheets Feeder Database",
+    "type": ["feeder", "database"],
+    "entry_point": "gsheet_feeder_db::GsheetsFeederDB",
    "requires_setup": True,
    "dependencies": {
        "python": ["loguru", "gspread", "slugify"],
@@ -15,7 +15,8 @@
        "header": {"default": 1, "help": "index of the header row (starts at 1)", "type": "int"},
        "service_account": {
            "default": "secrets/service_account.json",
-            "help": "service account JSON file path",
+            "help": "service account JSON file path. Learn how to create one: https://gspread.readthedocs.io/en/latest/oauth2.html",
+            "required": True,
        },
        "columns": {
            "default": {
@@ -34,16 +35,16 @@
                "wacz": "wacz",
                "replaywebpage": "replaywebpage",
            },
-            "help": "names of columns in the google sheet (stringified JSON object)",
-            "type": "auto_archiver.utils.json_loader",
+            "help": "Custom names for the columns in your Google sheet. If you don't want to use the default column names, change them with this setting",
+            "type": "json_loader",
        },
        "allow_worksheets": {
            "default": set(),
-            "help": "(CSV) only worksheets whose name is included in allow are included (overrides worksheet_block), leave empty so all are allowed",
+            "help": "A list of worksheet names that should be processed (overrides worksheet_block), leave empty so all are allowed",
        },
        "block_worksheets": {
            "default": set(),
-            "help": "(CSV) explicitly block some worksheets from being processed",
+            "help": "A list of worksheet names for worksheets that should be explicitly blocked from being processed",
        },
        "use_sheet_names_in_stored_paths": {
            "default": True,
@@ -52,8 +53,8 @@
        },
    },
    "description": """
-    GsheetsFeeder 
-    A Google Sheets-based feeder for the Auto Archiver.
+    GsheetsFeederDatabase
+    A Google Sheets-based feeder and optional database for the Auto Archiver.

    This reads data from Google Sheets and filters rows based on user-defined rules.
    The filtered rows are processed into `Metadata` objects.
@@ -63,9 +64,16 @@
    - Processes only worksheets allowed by the `allow_worksheets` and `block_worksheets` configurations.
    - Ensures only rows with valid URLs and unprocessed statuses are included for archival.
    - Supports organizing stored files into folder paths based on sheet and worksheet names.
+    - If the database is enabled, this updates the Google Sheet with the status of the archived URLs, including in progress, success or failure, and method used.
+    - Saves metadata such as title, text, timestamp, hashes, screenshots, and media URLs to designated columns.
+    - Formats media-specific metadata, such as thumbnails and PDQ hashes for the sheet.
+    - Skips redundant updates for empty or invalid data fields.

-    ### Notes
-    - Requires a Google Service Account JSON file for authentication. Suggested location is `secrets/gsheets_service_account.json`.
-    - Create the sheet using the template provided in the docs.
+    ### Setup
+    - Requires a Google Service Account JSON file for authentication, which should be stored in `secrets/gsheets_service_account.json`.
+    To set up a service account, follow the instructions [here](https://gspread.readthedocs.io/en/latest/oauth2.html).
+    - Define the `sheet` or `sheet_id` configuration to specify the sheet to archive.
+    - Customize the column names in your Google sheet using the `columns` configuration.
+    - The Google Sheet can be used soley as a feeder or as a feeder and database, but note you can't currently feed into the database from an alternate feeder.
    """,
 }
--- a/src/auto_archiver/modules/gsheet_feeder_db/gsheet_feeder_db.py
+++ b/src/auto_archiver/modules/gsheet_feeder_db/gsheet_feeder_db.py
@@ -0,0 +1,198 @@
+"""
+GsheetsFeeder: A Google Sheets-based feeder for the Auto Archiver.
+
+This reads data from Google Sheets and filters rows based on user-defined rules.
+The filtered rows are processed into `Metadata` objects.
+
+### Key properties
+- validates the sheet's structure and filters rows based on input configurations.
+- Ensures only rows with valid URLs and unprocessed statuses are included.
+"""
+
+import os
+from typing import Tuple, Union
+from urllib.parse import quote
+
+import gspread
+from loguru import logger
+from slugify import slugify
+
+from auto_archiver.core import Feeder, Database, Media
+from auto_archiver.core import Metadata
+from auto_archiver.modules.gsheet_feeder_db import GWorksheet
+from auto_archiver.utils.misc import get_current_timestamp
+
+
+class GsheetsFeederDB(Feeder, Database):
+    def setup(self) -> None:
+        self.gsheets_client = gspread.service_account(filename=self.service_account)
+        # TODO mv to validators
+        if not self.sheet and not self.sheet_id:
+            raise ValueError("You need to define either a 'sheet' name or a 'sheet_id' in your manifest.")
+
+    def open_sheet(self):
+        if self.sheet:
+            return self.gsheets_client.open(self.sheet)
+        else:  # self.sheet_id
+            return self.gsheets_client.open_by_key(self.sheet_id)
+
+    def __iter__(self) -> Metadata:
+        sh = self.open_sheet()
+        for ii, worksheet in enumerate(sh.worksheets()):
+            if not self.should_process_sheet(worksheet.title):
+                logger.debug(f"SKIPPED worksheet '{worksheet.title}' due to allow/block rules")
+                continue
+            logger.info(f"Opening worksheet {ii=}: {worksheet.title=} header={self.header}")
+            gw = GWorksheet(worksheet, header_row=self.header, columns=self.columns)
+            if len(missing_cols := self.missing_required_columns(gw)):
+                logger.warning(
+                    f"SKIPPED worksheet '{worksheet.title}' due to missing required column(s) for {missing_cols}"
+                )
+                continue
+
+            # process and yield metadata here:
+            yield from self._process_rows(gw)
+            logger.success(f"Finished worksheet {worksheet.title}")
+
+    def _process_rows(self, gw: GWorksheet):
+        for row in range(1 + self.header, gw.count_rows() + 1):
+            url = gw.get_cell(row, "url").strip()
+            if not len(url):
+                continue
+            original_status = gw.get_cell(row, "status")
+            status = gw.get_cell(row, "status", fresh=original_status in ["", None])
+            # TODO: custom status parser(?) aka should_retry_from_status
+            if status not in ["", None]:
+                continue
+
+            # All checks done - archival process starts here
+            m = Metadata().set_url(url)
+            self._set_context(m, gw, row)
+            yield m
+
+    def _set_context(self, m: Metadata, gw: GWorksheet, row: int) -> Metadata:
+        # TODO: Check folder value not being recognised
+        m.set_context("gsheet", {"row": row, "worksheet": gw})
+
+        if gw.get_cell_or_default(row, "folder", "") is None:
+            folder = ""
+        else:
+            folder = slugify(gw.get_cell_or_default(row, "folder", "").strip())
+        if len(folder):
+            if self.use_sheet_names_in_stored_paths:
+                m.set_context("folder", os.path.join(folder, slugify(self.sheet), slugify(gw.wks.title)))
+            else:
+                m.set_context("folder", folder)
+
+    def should_process_sheet(self, sheet_name: str) -> bool:
+        if len(self.allow_worksheets) and sheet_name not in self.allow_worksheets:
+            # ALLOW rules exist AND sheet name not explicitly allowed
+            return False
+        if len(self.block_worksheets) and sheet_name in self.block_worksheets:
+            # BLOCK rules exist AND sheet name is blocked
+            return False
+        return True
+
+    def missing_required_columns(self, gw: GWorksheet) -> list:
+        missing = []
+        for required_col in ["url", "status"]:
+            if not gw.col_exists(required_col):
+                missing.append(required_col)
+        return missing
+
+    def started(self, item: Metadata) -> None:
+        logger.warning(f"STARTED {item}")
+        gw, row = self._retrieve_gsheet(item)
+        gw.set_cell(row, "status", "Archive in progress")
+
+    def failed(self, item: Metadata, reason: str) -> None:
+        logger.error(f"FAILED {item}")
+        self._safe_status_update(item, f"Archive failed {reason}")
+
+    def aborted(self, item: Metadata) -> None:
+        logger.warning(f"ABORTED {item}")
+        self._safe_status_update(item, "")
+
+    def fetch(self, item: Metadata) -> Union[Metadata, bool]:
+        """check if the given item has been archived already"""
+        return False
+
+    def done(self, item: Metadata, cached: bool = False) -> None:
+        """archival result ready - should be saved to DB"""
+        logger.success(f"DONE {item.get_url()}")
+        gw, row = self._retrieve_gsheet(item)
+        # self._safe_status_update(item, 'done')
+
+        cell_updates = []
+        row_values = gw.get_row(row)
+
+        def batch_if_valid(col, val, final_value=None):
+            final_value = final_value or val
+            try:
+                if val and gw.col_exists(col) and gw.get_cell(row_values, col) == "":
+                    cell_updates.append((row, col, final_value))
+            except Exception as e:
+                logger.error(f"Unable to batch {col}={final_value} due to {e}")
+
+        status_message = item.status
+        if cached:
+            status_message = f"[cached] {status_message}"
+        cell_updates.append((row, "status", status_message))
+
+        media: Media = item.get_final_media()
+        if hasattr(media, "urls"):
+            batch_if_valid("archive", "\n".join(media.urls))
+        batch_if_valid("date", True, get_current_timestamp())
+        batch_if_valid("title", item.get_title())
+        batch_if_valid("text", item.get("content", ""))
+        batch_if_valid("timestamp", item.get_timestamp())
+        if media:
+            batch_if_valid("hash", media.get("hash", "not-calculated"))
+
+        # merge all pdq hashes into a single string, if present
+        pdq_hashes = []
+        all_media = item.get_all_media()
+        for m in all_media:
+            if pdq := m.get("pdq_hash"):
+                pdq_hashes.append(pdq)
+        if len(pdq_hashes):
+            batch_if_valid("pdq_hash", ",".join(pdq_hashes))
+
+        if (screenshot := item.get_media_by_id("screenshot")) and hasattr(screenshot, "urls"):
+            batch_if_valid("screenshot", "\n".join(screenshot.urls))
+
+        if thumbnail := item.get_first_image("thumbnail"):
+            if hasattr(thumbnail, "urls"):
+                batch_if_valid("thumbnail", f'=IMAGE("{thumbnail.urls[0]}")')
+
+        if browsertrix := item.get_media_by_id("browsertrix"):
+            batch_if_valid("wacz", "\n".join(browsertrix.urls))
+            batch_if_valid(
+                "replaywebpage",
+                "\n".join(
+                    [
+                        f"https://replayweb.page/?source={quote(wacz)}#view=pages&url={quote(item.get_url())}"
+                        for wacz in browsertrix.urls
+                    ]
+                ),
+            )
+
+        gw.batch_set_cell(cell_updates)
+
+    def _safe_status_update(self, item: Metadata, new_status: str) -> None:
+        try:
+            gw, row = self._retrieve_gsheet(item)
+            gw.set_cell(row, "status", new_status)
+        except Exception as e:
+            logger.debug(f"Unable to update sheet: {e}")
+
+    def _retrieve_gsheet(self, item: Metadata) -> Tuple[GWorksheet, int]:
+        if gsheet := item.get_context("gsheet"):
+            gw: GWorksheet = gsheet.get("worksheet")
+            row: int = gsheet.get("row")
+        elif self.sheet_id:
+            logger.error(
+                f"Unable to retrieve Gsheet for {item.get_url()}, GsheetDB must be used alongside GsheetFeeder."
+            )
+
+        return gw, row
--- a/src/auto_archiver/modules/gsheet_feeder_db/gworksheet.py
+++ b/src/auto_archiver/modules/gsheet_feeder_db/gworksheet.py
@@ -5,23 +5,25 @@ class GWorksheet:
    """
    This class makes read/write operations to the a worksheet easier.
    It can read the headers from a custom row number, but the row references
-    should always include the offset of the header. 
-    eg: if header=4, row 5 will be the first with data. 
+    should always include the offset of the header.
+    eg: if header=4, row 5 will be the first with data.
    """
+
    COLUMN_NAMES = {
-        'url': 'link',
-        'status': 'archive status',
-        'folder': 'destination folder',
-        'archive': 'archive location',
-        'date': 'archive date',
-        'thumbnail': 'thumbnail',
-        'timestamp': 'upload timestamp',
-        'title': 'upload title',
-        'screenshot': 'screenshot',
-        'hash': 'hash',
-        'pdq_hash': 'perceptual hashes',
-        'wacz': 'wacz',
-        'replaywebpage': 'replaywebpage',
+        "url": "link",
+        "status": "archive status",
+        "folder": "destination folder",
+        "archive": "archive location",
+        "date": "archive date",
+        "thumbnail": "thumbnail",
+        "timestamp": "upload timestamp",
+        "title": "upload title",
+        "text": "text content",
+        "screenshot": "screenshot",
+        "hash": "hash",
+        "pdq_hash": "perceptual hashes",
+        "wacz": "wacz",
+        "replaywebpage": "replaywebpage",
    }

    def __init__(self, worksheet, columns=COLUMN_NAMES, header_row=1):
@@ -35,7 +37,7 @@ class GWorksheet:

    def _check_col_exists(self, col: str):
        if col not in self.columns:
-            raise Exception(f'Column {col} is not in the configured column names: {self.columns.keys()}')
+            raise Exception(f"Column {col} is not in the configured column names: {self.columns.keys()}")

    def _col_index(self, col: str):
        self._check_col_exists(col)
@@ -57,7 +59,7 @@ class GWorksheet:

    def get_cell(self, row, col: str, fresh=False):
        """
-        returns the cell value from (row, col), 
+        returns the cell value from (row, col),
        where row can be an index (1-based) OR list of values
        as received from self.get_row(row)
        if fresh=True, the sheet is queried again for this cell
@@ -66,11 +68,11 @@ class GWorksheet:

        if fresh:
            return self.wks.cell(row, col_index + 1).value
-        if type(row) == int:
+        if isinstance(row, int):
            row = self.get_row(row)

        if col_index >= len(row):
-            return ''
+            return ""
        return row[col_index]

    def get_cell_or_default(self, row, col: str, default: str = None, fresh=False, when_empty_use_default=True):
@@ -82,7 +84,7 @@ class GWorksheet:
            if when_empty_use_default and val.strip() == "":
                return default
            return val
-        except:
+        except Exception:
            return default

    def set_cell(self, row: int, col: str, val):
@@ -95,13 +97,9 @@ class GWorksheet:
        receives a list of [(row:int, col:str, val)] and batch updates it, the parameters are the same as in the self.set_cell() method
        """
        cell_updates = [
-            {
-                'range': self.to_a1(row, col),
-                'values': [[str(val)[0:49999]]]
-            }
-            for row, col, val in cell_updates
+            {"range": self.to_a1(row, col), "values": [[str(val)[0:49999]]]} for row, col, val in cell_updates
        ]
-        self.wks.batch_update(cell_updates, value_input_option='USER_ENTERED')
+        self.wks.batch_update(cell_updates, value_input_option="USER_ENTERED")

    def to_a1(self, row: int, col: str):
        # row is 1-based
--- a/src/auto_archiver/modules/hash_enricher/init.py
+++ b/src/auto_archiver/modules/hash_enricher/init.py
@@ -1 +1 @@
-from .hash_enricher import HashEnricher
+from .hash_enricher import HashEnricher
--- a/src/auto_archiver/modules/hash_enricher/manifest.py
+++ b/src/auto_archiver/modules/hash_enricher/manifest.py
@@ -3,16 +3,17 @@
    "type": ["enricher"],
    "requires_setup": False,
    "dependencies": {
-                          "python": ["loguru"],
+        "python": ["loguru"],
    },
    "configs": {
-            "algorithm": {"default": "SHA-256", "help": "hash algorithm to use", "choices": ["SHA-256", "SHA3-512"]},
-            # TODO add non-negative requirement to match previous implementation?
-            "chunksize": {"default": 16000000,
-                          "help": "number of bytes to use when reading files in chunks (if this value is too large you will run out of RAM), default is 16MB",
-                          'type': 'int',
-                          },
+        "algorithm": {"default": "SHA-256", "help": "hash algorithm to use", "choices": ["SHA-256", "SHA3-512"]},
+        # TODO add non-negative requirement to match previous implementation?
+        "chunksize": {
+            "default": 16000000,
+            "help": "number of bytes to use when reading files in chunks (if this value is too large you will run out of RAM), default is 16MB",
+            "type": "int",
        },
+    },
    "description": """
 Generates cryptographic hashes for media files to ensure data integrity and authenticity.

--- a/src/auto_archiver/modules/hash_enricher/hash_enricher.py
+++ b/src/auto_archiver/modules/hash_enricher/hash_enricher.py
@@ -1,4 +1,4 @@
-""" Hash Enricher for generating cryptographic hashes of media files.
+"""Hash Enricher for generating cryptographic hashes of media files.

 The `HashEnricher` calculates cryptographic hashes (e.g., SHA-256, SHA3-512)
 for media files stored in `Metadata` objects. These hashes are used for
@@ -7,6 +7,7 @@ exact duplicates. The hash is computed by reading the file's bytes in chunks,
 making it suitable for handling large files efficiently.

 """
+
 import hashlib
 from loguru import logger

@@ -20,7 +21,6 @@ class HashEnricher(Enricher):
    Calculates hashes for Media instances
    """

-
    def enrich(self, to_enrich: Metadata) -> None:
        url = to_enrich.get_url()
        logger.debug(f"calculating media hashes for {url=} (using {self.algorithm})")
@@ -35,5 +35,6 @@ class HashEnricher(Enricher):
            hash_algo = hashlib.sha256
        elif self.algorithm == "SHA3-512":
            hash_algo = hashlib.sha3_512
-        else: return ""
+        else:
+            return ""
        return calculate_file_hash(filename, hash_algo, self.chunksize)
--- a/src/auto_archiver/modules/html_formatter/init.py
+++ b/src/auto_archiver/modules/html_formatter/init.py
@@ -1 +1 @@
-from .html_formatter import HtmlFormatter
+from .html_formatter import HtmlFormatter
--- a/src/auto_archiver/modules/html_formatter/manifest.py
+++ b/src/auto_archiver/modules/html_formatter/manifest.py
@@ -2,12 +2,13 @@
    "name": "HTML Formatter",
    "type": ["formatter"],
    "requires_setup": False,
-    "dependencies": {
-                          "python": ["hash_enricher", "loguru", "jinja2"],
-                          "bin": [""]
-    },
+    "dependencies": {"python": ["hash_enricher", "loguru", "jinja2"], "bin": [""]},
    "configs": {
-            "detect_thumbnails": {"default": True, "help": "if true will group by thumbnails generated by thumbnail enricher by id 'thumbnail_00'"}
+        "detect_thumbnails": {
+            "default": True,
+            "help": "if true will group by thumbnails generated by thumbnail enricher by id 'thumbnail_00'",
+            "type": "bool",
        },
+    },
    "description": """ """,
 }
--- a/src/auto_archiver/modules/html_formatter/html_formatter.py
+++ b/src/auto_archiver/modules/html_formatter/html_formatter.py
@@ -1,5 +1,7 @@
 from __future__ import annotations
-import mimetypes, os, pathlib
+import mimetypes
+import os
+import pathlib
 from jinja2 import Environment, FileSystemLoader
 from urllib.parse import quote
 from loguru import logger
@@ -11,6 +13,7 @@ from auto_archiver.core import Metadata, Media
 from auto_archiver.core import Formatter
 from auto_archiver.utils.misc import random_str

+
 class HtmlFormatter(Formatter):
    environment: Environment = None
    template: any = None
@@ -21,9 +24,9 @@ class HtmlFormatter(Formatter):
        self.environment = Environment(loader=FileSystemLoader(template_dir), autoescape=True)

        # JinjaHelper class static methods are added as filters
-        self.environment.filters.update({
-            k: v.__func__ for k, v in JinjaHelpers.__dict__.items() if isinstance(v, staticmethod)
-        })
+        self.environment.filters.update(
+            {k: v.__func__ for k, v in JinjaHelpers.__dict__.items() if isinstance(v, staticmethod)}
+        )

        # Load a specific template or default to "html_template.html"
        template_name = self.config.get("template_name", "html_template.html")
@@ -36,11 +39,7 @@ class HtmlFormatter(Formatter):
            return

        content = self.template.render(
-            url=url,
-            title=item.get_title(),
-            media=item.media,
-            metadata=item.metadata,
-            version=__version__
+            url=url, title=item.get_title(), media=item.media, metadata=item.metadata, version=__version__
        )

        html_path = os.path.join(self.tmp_dir, f"formatted{random_str(24)}.html")
@@ -49,7 +48,7 @@ class HtmlFormatter(Formatter):
        final_media = Media(filename=html_path, _mimetype="text/html")

        # get the already instantiated hash_enricher module
-        he = self.module_factory.get_module('hash_enricher', self.config)
+        he = self.module_factory.get_module("hash_enricher", self.config)
        if len(hd := he.calculate_hash(final_media.filename)):
            final_media.set("hash", f"{he.algorithm}:{hd}")

--- a/src/auto_archiver/modules/instagram_api_extractor/manifest.py
+++ b/src/auto_archiver/modules/instagram_api_extractor/manifest.py
@@ -2,18 +2,18 @@
    "name": "Instagram API Extractor",
    "type": ["extractor"],
    "entry_point": "instagram_api_extractor::InstagramAPIExtractor",
-    "dependencies":
-        {"python": ["requests",
-                    "loguru",
-                    "retrying",
-                    "tqdm",],
-         },
+    "dependencies": {
+        "python": [
+            "requests",
+            "loguru",
+            "retrying",
+            "tqdm",
+        ],
+    },
    "requires_setup": True,
    "configs": {
-        "access_token": {"default": None,
-                         "help": "a valid instagrapi-api token"},
-        "api_endpoint": {"required": True,
-                         "help": "API endpoint to use"},
+        "access_token": {"default": None, "help": "a valid instagrapi-api token"},
+        "api_endpoint": {"required": True, "help": "API endpoint to use"},
        "full_profile": {
            "default": False,
            "type": "bool",
--- a/src/auto_archiver/modules/instagram_api_extractor/instagram_api_extractor.py
+++ b/src/auto_archiver/modules/instagram_api_extractor/instagram_api_extractor.py
@@ -36,21 +36,16 @@ class InstagramAPIExtractor(Extractor):
        if self.api_endpoint[-1] == "/":
            self.api_endpoint = self.api_endpoint[:-1]

-
    def download(self, item: Metadata) -> Metadata:
        url = item.get_url()

-        url.replace("instagr.com", "instagram.com").replace(
-            "instagr.am", "instagram.com"
-        )
+        url.replace("instagr.com", "instagram.com").replace("instagr.am", "instagram.com")
        insta_matches = self.valid_url.findall(url)
        logger.info(f"{insta_matches=}")
        if not len(insta_matches) or len(insta_matches[0]) != 3:
            return
        if len(insta_matches) > 1:
-            logger.warning(
-                f"Multiple instagram matches found in {url=}, using the first one"
-            )
+            logger.warning(f"Multiple instagram matches found in {url=}, using the first one")
            return
        g1, g2, g3 = insta_matches[0][0], insta_matches[0][1], insta_matches[0][2]
        if g1 == "":
@@ -73,23 +68,20 @@ class InstagramAPIExtractor(Extractor):
    def call_api(self, path: str, params: dict) -> dict:
        headers = {"accept": "application/json", "x-access-key": self.access_token}
        logger.debug(f"calling {self.api_endpoint}/{path} with {params=}")
-        return requests.get(
-            f"{self.api_endpoint}/{path}", headers=headers, params=params
-        ).json()
+        return requests.get(f"{self.api_endpoint}/{path}", headers=headers, params=params).json()

    def cleanup_dict(self, d: dict | list) -> dict:
        # repeats 3 times to remove nested empty values
        if not self.minimize_json_output:
            return d
-        if type(d) == list:
+        if isinstance(d, list):
            return [self.cleanup_dict(v) for v in d]
-        if type(d) != dict:
+        if not isinstance(d, dict):
            return d
        return {
            k: clean_v
            for k, v in d.items()
-            if (clean_v := self.cleanup_dict(v))
-            not in [0.0, 0, [], {}, "", None, "null"]
+            if (clean_v := self.cleanup_dict(v)) not in [0.0, 0, [], {}, "", None, "null"]
            and k not in ["x", "y", "width", "height"]
        }

@@ -103,7 +95,7 @@ class InstagramAPIExtractor(Extractor):
        result.set_title(user.get("full_name", username)).set("data", user)
        if pic_url := user.get("profile_pic_url_hd", user.get("profile_pic_url")):
            filename = self.download_from_url(pic_url)
-            result.add_media(Media(filename=filename), id=f"profile_picture")
+            result.add_media(Media(filename=filename), id="profile_picture")

        if self.full_profile:
            user_id = user.get("pk")
@@ -126,9 +118,7 @@ class InstagramAPIExtractor(Extractor):
            try:
                self.download_all_tagged(result, user_id)
            except Exception as e:
-                result.append(
-                    "errors", f"Error downloading tagged posts for {username}"
-                )
+                result.append("errors", f"Error downloading tagged posts for {username}")
                logger.error(f"Error downloading tagged posts for {username}: {e}")

            # download all highlights
@@ -143,7 +133,7 @@ class InstagramAPIExtractor(Extractor):

    def download_all_highlights(self, result, username, user_id):
        count_highlights = 0
-        highlights = self.call_api(f"v1/user/highlights", {"user_id": user_id})
+        highlights = self.call_api("v1/user/highlights", {"user_id": user_id})
        for h in highlights:
            try:
                h_info = self._download_highlights_reusable(result, h.get("pk"))
@@ -153,26 +143,17 @@ class InstagramAPIExtractor(Extractor):
                    "errors",
                    f"Error downloading highlight id{h.get('pk')} for {username}",
                )
-                logger.error(
-                    f"Error downloading highlight id{h.get('pk')} for {username}: {e}"
-                )
-            if (
-                self.full_profile_max_posts
-                and count_highlights >= self.full_profile_max_posts
-            ):
-                logger.info(
-                    f"HIGHLIGHTS reached full_profile_max_posts={self.full_profile_max_posts}"
-                )
+                logger.error(f"Error downloading highlight id{h.get('pk')} for {username}: {e}")
+            if self.full_profile_max_posts and count_highlights >= self.full_profile_max_posts:
+                logger.info(f"HIGHLIGHTS reached full_profile_max_posts={self.full_profile_max_posts}")
                break
        result.set("#highlights", count_highlights)

-    def download_post(
-        self, result: Metadata, code: str = None, id: str = None, context: str = None
-    ) -> Metadata:
+    def download_post(self, result: Metadata, code: str = None, id: str = None, context: str = None) -> Metadata:
        if id:
-            post = self.call_api(f"v1/media/by/id", {"id": id})
+            post = self.call_api("v1/media/by/id", {"id": id})
        else:
-            post = self.call_api(f"v1/media/by/code", {"code": code})
+            post = self.call_api("v1/media/by/code", {"code": code})
        assert post, f"Post {id or code} not found"

        if caption_text := post.get("caption_text"):
@@ -192,15 +173,11 @@ class InstagramAPIExtractor(Extractor):
        return result.success("insta highlights")

    def _download_highlights_reusable(self, result: Metadata, id: str) -> dict:
-        full_h = self.call_api(f"v2/highlight/by/id", {"id": id})
+        full_h = self.call_api("v2/highlight/by/id", {"id": id})
        h_info = full_h.get("response", {}).get("reels", {}).get(f"highlight:{id}")
        assert h_info, f"Highlight {id} not found: {full_h=}"

-        if (
-            cover_media := h_info.get("cover_media", {})
-            .get("cropped_image_version", {})
-            .get("url")
-        ):
+        if cover_media := h_info.get("cover_media", {}).get("cropped_image_version", {}).get("url"):
            filename = self.download_from_url(cover_media)
            result.add_media(Media(filename=filename), id=f"cover_media highlight {id}")

@@ -210,9 +187,7 @@ class InstagramAPIExtractor(Extractor):
                self.scrape_item(result, h, "highlight")
            except Exception as e:
                result.append("errors", f"Error downloading highlight {h.get('id')}")
-                logger.error(
-                    f"Error downloading highlight, skipping {h.get('id')}: {e}"
-                )
+                logger.error(f"Error downloading highlight, skipping {h.get('id')}: {e}")

        return h_info

@@ -225,7 +200,7 @@ class InstagramAPIExtractor(Extractor):
        return result.success(f"insta stories {now}")

    def _download_stories_reusable(self, result: Metadata, username: str) -> list[dict]:
-        stories = self.call_api(f"v1/user/stories/by/username", {"username": username})
+        stories = self.call_api("v1/user/stories/by/username", {"username": username})
        if not stories or not len(stories):
            return []
        stories = stories[::-1]  # newest to oldest
@@ -244,10 +219,8 @@ class InstagramAPIExtractor(Extractor):

        post_count = 0
        while end_cursor != "":
-            posts = self.call_api(
-                f"v1/user/medias/chunk", {"user_id": user_id, "end_cursor": end_cursor}
-            )
-            if not len(posts) or not type(posts) == list or len(posts) != 2:
+            posts = self.call_api("v1/user/medias/chunk", {"user_id": user_id, "end_cursor": end_cursor})
+            if not posts or not isinstance(posts, list) or len(posts) != 2:
                break
            posts, end_cursor = posts[0], posts[1]
            logger.info(f"parsing {len(posts)} posts, next {end_cursor=}")
@@ -260,13 +233,8 @@ class InstagramAPIExtractor(Extractor):
                    logger.error(f"Error downloading post, skipping {p.get('id')}: {e}")
                pbar.update(1)
                post_count += 1
-            if (
-                self.full_profile_max_posts
-                and post_count >= self.full_profile_max_posts
-            ):
-                logger.info(
-                    f"POSTS reached full_profile_max_posts={self.full_profile_max_posts}"
-                )
+            if self.full_profile_max_posts and post_count >= self.full_profile_max_posts:
+                logger.info(f"POSTS reached full_profile_max_posts={self.full_profile_max_posts}")
                break
        result.set("#posts", post_count)

@@ -275,10 +243,8 @@ class InstagramAPIExtractor(Extractor):
        pbar = tqdm(desc="downloading tagged posts")

        tagged_count = 0
-        while next_page_id != None:
-            resp = self.call_api(
-                f"v2/user/tag/medias", {"user_id": user_id, "page_id": next_page_id}
-            )
+        while next_page_id is not None:
+            resp = self.call_api("v2/user/tag/medias", {"user_id": user_id, "page_id": next_page_id})
            posts = resp.get("response", {}).get("items", [])
            if not len(posts):
                break
@@ -290,21 +256,12 @@ class InstagramAPIExtractor(Extractor):
                try:
                    self.scrape_item(result, p, "tagged")
                except Exception as e:
-                    result.append(
-                        "errors", f"Error downloading tagged post {p.get('id')}"
-                    )
-                    logger.error(
-                        f"Error downloading tagged post, skipping {p.get('id')}: {e}"
-                    )
+                    result.append("errors", f"Error downloading tagged post {p.get('id')}")
+                    logger.error(f"Error downloading tagged post, skipping {p.get('id')}: {e}")
                pbar.update(1)
                tagged_count += 1
-            if (
-                self.full_profile_max_posts
-                and tagged_count >= self.full_profile_max_posts
-            ):
-                logger.info(
-                    f"TAGS reached full_profile_max_posts={self.full_profile_max_posts}"
-                )
+            if self.full_profile_max_posts and tagged_count >= self.full_profile_max_posts:
+                logger.info(f"TAGS reached full_profile_max_posts={self.full_profile_max_posts}")
                break
        result.set("#tagged", tagged_count)

@@ -318,9 +275,7 @@ class InstagramAPIExtractor(Extractor):
        context can be used to give specific id prefixes to media
        """
        if "clips_metadata" in item:
-            if reusable_text := item.get("clips_metadata", {}).get(
-                "reusable_text_attribute_string"
-            ):
+            if reusable_text := item.get("clips_metadata", {}).get("reusable_text_attribute_string"):
                item["clips_metadata_text"] = reusable_text
            if self.minimize_json_output:
                del item["clips_metadata"]
--- a/src/auto_archiver/modules/instagram_extractor/init.py
+++ b/src/auto_archiver/modules/instagram_extractor/init.py
@@ -1 +1 @@
-from .instagram_extractor import InstagramExtractor
+from .instagram_extractor import InstagramExtractor
--- a/src/auto_archiver/modules/instagram_extractor/manifest.py
+++ b/src/auto_archiver/modules/instagram_extractor/manifest.py
@@ -9,26 +9,30 @@
    },
    "requires_setup": True,
    "configs": {
-        "username": {"required": True,
-                     "help": "a valid Instagram username"},
+        "username": {"required": True, "help": "A valid Instagram username."},
        "password": {
            "required": True,
-            "help": "the corresponding Instagram account password",
+            "help": "The corresponding Instagram account password.",
        },
        "download_folder": {
            "default": "instaloader",
-            "help": "name of a folder to temporarily download content to",
+            "help": "Name of a folder to temporarily download content to.",
        },
        "session_file": {
            "default": "secrets/instaloader.session",
-            "help": "path to the instagram session which saves session credentials",
+            "help": "Path to the instagram session file which saves session credentials. If one doesn't exist this gives the path to store a new one.",
        },
        # TODO: fine-grain
        # "download_stories": {"default": True, "help": "if the link is to a user profile: whether to get stories information"},
    },
    "description": """
-    Uses the [Instaloader library](https://instaloader.github.io/as-module.html) to download content from Instagram. This class handles both individual posts
-    and user profiles, downloading as much information as possible, including images, videos, text, stories,
+    Uses the [Instaloader library](https://instaloader.github.io/as-module.html) to download content from Instagram. 
+    
+      > ⚠️ **Warning**  
+      > This module is not actively maintained due to known issues with blocking.  
+      > Prioritise usage of the [Instagram Tbot Extractor](./instagram_tbot_extractor.md) and [Instagram API Extractor](./instagram_api_extractor.md)
+  
+    This class handles both individual posts and user profiles, downloading as much information as possible, including images, videos, text, stories,
    highlights, and tagged posts. 
    Authentication is required via username/password or a session file.
                    
--- a/src/auto_archiver/modules/instagram_extractor/instagram_extractor.py
+++ b/src/auto_archiver/modules/instagram_extractor/instagram_extractor.py
@@ -1,9 +1,12 @@
-""" Uses the Instaloader library to download content from Instagram. This class handles both individual posts
-    and user profiles, downloading as much information as possible, including images, videos, text, stories,
-    highlights, and tagged posts. Authentication is required via username/password or a session file.
+"""Uses the Instaloader library to download content from Instagram. This class handles both individual posts
+and user profiles, downloading as much information as possible, including images, videos, text, stories,
+highlights, and tagged posts. Authentication is required via username/password or a session file.

 """
-import re, os, shutil, traceback
+
+import re
+import os
+import shutil
 import instaloader
 from loguru import logger

@@ -11,14 +14,14 @@ from auto_archiver.core import Extractor
 from auto_archiver.core import Metadata
 from auto_archiver.core import Media

+
 class InstagramExtractor(Extractor):
    """
    Uses Instaloader to download either a post (inc images, videos, text) or as much as possible from a profile (posts, stories, highlights, ...)
    """
+
    # NB: post regex should be tested before profile
-
    valid_url = re.compile(r"(?:(?:http|https):\/\/)?(?:www.)?(?:instagram.com|instagr.am|instagr.com)\/")
-
    # https://regex101.com/r/MGPquX/1
    post_pattern = re.compile(r"{valid_url}(?:p|reel)\/(\w+)".format(valid_url=valid_url))
    # https://regex101.com/r/6Wbsxa/1
@@ -26,22 +29,23 @@ class InstagramExtractor(Extractor):
    # TODO: links to stories

    def setup(self) -> None:
-
        self.insta = instaloader.Instaloader(
-            download_geotags=True, download_comments=True, compress_json=False, dirname_pattern=self.download_folder, filename_pattern="{date_utc}_UTC_{target}__{typename}"
+            download_geotags=True,
+            download_comments=True,
+            compress_json=False,
+            dirname_pattern=self.download_folder,
+            filename_pattern="{date_utc}_UTC_{target}__{typename}",
        )
        try:
            self.insta.load_session_from_file(self.username, self.session_file)
-        except Exception as e:
-            logger.error(f"Unable to login from session file: {e}\n{traceback.format_exc()}")
+        except Exception:
            try:
-                self.insta.login(self.username, config.instagram_self.password)
-                # TODO: wait for this issue to be fixed https://github.com/instaloader/instaloader/issues/1758
+                logger.debug("Session file failed", exc_info=True)
+                logger.info("No valid session file found - Attempting login with use and password.")
+                self.insta.login(self.username, self.password)
                self.insta.save_session_to_file(self.session_file)
-            except Exception as e2:
-                logger.error(f"Unable to finish login (retrying from file): {e2}\n{traceback.format_exc()}")
-
-
+            except Exception as e:
+                logger.error(f"Failed to setup Instagram Extractor with Instagrapi. {e}")

    def download(self, item: Metadata) -> Metadata:
        url = item.get_url()
@@ -51,7 +55,8 @@ class InstagramExtractor(Extractor):
        profile_matches = self.profile_pattern.findall(url)

        # return if not a valid instagram link
-        if not len(post_matches) and not len(profile_matches): return
+        if not len(post_matches) and not len(profile_matches):
+            return

        result = None
        try:
@@ -63,7 +68,9 @@ class InstagramExtractor(Extractor):
            elif len(profile_matches):
                result = self.download_profile(url, profile_matches[0])
        except Exception as e:
-            logger.error(f"Failed to download with instagram extractor due to: {e}, make sure your account credentials are valid.")
+            logger.error(
+                f"Failed to download with instagram extractor due to: {e}, make sure your account credentials are valid."
+            )
        finally:
            shutil.rmtree(self.download_folder, ignore_errors=True)
        return result
@@ -82,35 +89,50 @@ class InstagramExtractor(Extractor):
        profile = instaloader.Profile.from_username(self.insta.context, username)
        try:
            for post in profile.get_posts():
-                try: self.insta.download_post(post, target=f"profile_post_{post.owner_username}")
-                except Exception as e: logger.error(f"Failed to download post: {post.shortcode}: {e}")
-        except Exception as e: logger.error(f"Failed profile.get_posts: {e}")
+                try:
+                    self.insta.download_post(post, target=f"profile_post_{post.owner_username}")
+                except Exception as e:
+                    logger.error(f"Failed to download post: {post.shortcode}: {e}")
+        except Exception as e:
+            logger.error(f"Failed profile.get_posts: {e}")

        try:
            for post in profile.get_tagged_posts():
-                try: self.insta.download_post(post, target=f"tagged_post_{post.owner_username}")
-                except Exception as e: logger.error(f"Failed to download tagged post: {post.shortcode}: {e}")
-        except Exception as e: logger.error(f"Failed profile.get_tagged_posts: {e}")
+                try:
+                    self.insta.download_post(post, target=f"tagged_post_{post.owner_username}")
+                except Exception as e:
+                    logger.error(f"Failed to download tagged post: {post.shortcode}: {e}")
+        except Exception as e:
+            logger.error(f"Failed profile.get_tagged_posts: {e}")

        try:
            for post in profile.get_igtv_posts():
-                try: self.insta.download_post(post, target=f"igtv_post_{post.owner_username}")
-                except Exception as e: logger.error(f"Failed to download igtv post: {post.shortcode}: {e}")
-        except Exception as e: logger.error(f"Failed profile.get_igtv_posts: {e}")
+                try:
+                    self.insta.download_post(post, target=f"igtv_post_{post.owner_username}")
+                except Exception as e:
+                    logger.error(f"Failed to download igtv post: {post.shortcode}: {e}")
+        except Exception as e:
+            logger.error(f"Failed profile.get_igtv_posts: {e}")

        try:
            for story in self.insta.get_stories([profile.userid]):
                for item in story.get_items():
-                    try: self.insta.download_storyitem(item, target=f"story_item_{story.owner_username}")
-                    except Exception as e: logger.error(f"Failed to download story item: {item}: {e}")
-        except Exception as e: logger.error(f"Failed get_stories: {e}")
+                    try:
+                        self.insta.download_storyitem(item, target=f"story_item_{story.owner_username}")
+                    except Exception as e:
+                        logger.error(f"Failed to download story item: {item}: {e}")
+        except Exception as e:
+            logger.error(f"Failed get_stories: {e}")

        try:
            for highlight in self.insta.get_highlights(profile.userid):
                for item in highlight.get_items():
-                    try: self.insta.download_storyitem(item, target=f"highlight_item_{highlight.owner_username}")
-                    except Exception as e: logger.error(f"Failed to download highlight item: {item}: {e}")
-        except Exception as e: logger.error(f"Failed get_highlights: {e}")
+                    try:
+                        self.insta.download_storyitem(item, target=f"highlight_item_{highlight.owner_username}")
+                    except Exception as e:
+                        logger.error(f"Failed to download highlight item: {item}: {e}")
+        except Exception as e:
+            logger.error(f"Failed get_highlights: {e}")

        return self.process_downloads(url, f"@{username}", profile._asdict(), None)

@@ -122,7 +144,8 @@ class InstagramExtractor(Extractor):
            all_media = []
            for f in os.listdir(self.download_folder):
                if os.path.isfile((filename := os.path.join(self.download_folder, f))):
-                    if filename[-4:] == ".txt": continue
+                    if filename[-4:] == ".txt":
+                        continue
                    all_media.append(Media(filename))

            assert len(all_media) > 1, "No uploaded media found"
--- a/src/auto_archiver/modules/instagram_tbot_extractor/manifest.py
+++ b/src/auto_archiver/modules/instagram_tbot_extractor/manifest.py
@@ -1,16 +1,21 @@
 {
    "name": "Instagram Telegram Bot Extractor",
    "type": ["extractor"],
-    "dependencies": {"python": ["loguru", "telethon",],
-                              },
+    "dependencies": {
+        "python": [
+            "loguru",
+            "telethon",
+        ],
+    },
    "requires_setup": True,
    "configs": {
-            "api_id": {"default": None, "help": "telegram API_ID value, go to https://my.telegram.org/apps"},
-            "api_hash": {"default": None, "help": "telegram API_HASH value, go to https://my.telegram.org/apps"},
-            "session_file": {"default": "secrets/anon-insta", "help": "optional, records the telegram login session for future usage, '.session' will be appended to the provided value."},
-            "timeout": {"default": 45,
-                        "type": "int",
-                        "help": "timeout to fetch the instagram content in seconds."},
+        "api_id": {"default": None, "help": "telegram API_ID value, go to https://my.telegram.org/apps"},
+        "api_hash": {"default": None, "help": "telegram API_HASH value, go to https://my.telegram.org/apps"},
+        "session_file": {
+            "default": "secrets/anon-insta",
+            "help": "optional, records the telegram login session for future usage, '.session' will be appended to the provided value.",
+        },
+        "timeout": {"default": 45, "type": "int", "help": "timeout to fetch the instagram content in seconds."},
    },
    "description": """
 The `InstagramTbotExtractor` module uses a Telegram bot (`instagram_load_bot`) to fetch and archive Instagram content,
--- a/src/auto_archiver/modules/instagram_tbot_extractor/instagram_tbot_extractor.py
+++ b/src/auto_archiver/modules/instagram_tbot_extractor/instagram_tbot_extractor.py
@@ -51,7 +51,7 @@ class InstagramTbotExtractor(Extractor):
        """Initializes the Telegram client."""
        try:
            self.client = TelegramClient(self.session_file, self.api_id, self.api_hash)
-        except OperationalError as e:
+        except OperationalError:
            logger.error(
                f"Unable to access the {self.session_file} session. "
                "Ensure that you don't use the same session file here and in telethon_extractor. "
@@ -65,15 +65,15 @@ class InstagramTbotExtractor(Extractor):
        session_file_name = self.session_file + ".session"
        if os.path.exists(session_file_name):
            os.remove(session_file_name)
-        
+
    def download(self, item: Metadata) -> Metadata:
        url = item.get_url()
-        if not "instagram.com" in url: return False
+        if "instagram.com" not in url:
+            return False

        result = Metadata()
        tmp_dir = self.tmp_dir
        with self.client.start():
-
            chat, since_id = self._send_url_to_bot(url)
            message = self._process_messages(chat, since_id, tmp_dir, result)

@@ -104,19 +104,20 @@ class InstagramTbotExtractor(Extractor):
        message = ""
        time.sleep(3)
        # media is added before text by the bot so it can be used as a stop-logic mechanism
-        while attempts < (self.timeout - 3) and (not message or not len(seen_media)):
+        while attempts < max(self.timeout - 3, 3) and (not message or not len(seen_media)):
            attempts += 1
            time.sleep(1)
            for post in self.client.iter_messages(chat, min_id=since_id):
                since_id = max(since_id, post.id)
                # Skip known filler message:
-                if post.message == 'The bot receives information through https://hikerapi.com/p/hJqpppqi':
+                if post.message == "The bot receives information through https://hikerapi.com/p/hJqpppqi":
                    continue
                if post.media and post.id not in seen_media:
-                    filename_dest = os.path.join(tmp_dir, f'{chat.id}_{post.id}')
+                    filename_dest = os.path.join(tmp_dir, f"{chat.id}_{post.id}")
                    media = self.client.download_media(post.media, filename_dest)
                    if media:
                        result.add_media(Media(media))
                        seen_media.append(post.id)
-                if post.message: message += post.message
-        return message.strip()
+                if post.message:
+                    message += post.message
+        return message.strip()
--- a/src/auto_archiver/modules/local_storage/init.py
+++ b/src/auto_archiver/modules/local_storage/init.py
@@ -1 +1 @@
-from .local_storage import LocalStorage
+from .local_storage import LocalStorage
--- a/src/auto_archiver/modules/local_storage/manifest.py
+++ b/src/auto_archiver/modules/local_storage/manifest.py
@@ -13,11 +13,15 @@
        },
        "filename_generator": {
            "default": "static",
-            "help": "how to name stored files: 'random' creates a random string; 'static' uses a replicable strategy such as a hash.",
+            "help": "how to name stored files: 'random' creates a random string; 'static' uses a hash, with the settings of the 'hash_enricher' module (defaults to SHA256 if not enabled)",
            "choices": ["random", "static"],
        },
        "save_to": {"default": "./local_archive", "help": "folder where to save archived content"},
-        "save_absolute": {"default": True, "help": "whether the path to the stored file is absolute or relative in the output result inc. formatters (Warning: saving an absolute path will show your computer's file structure)"},
+        "save_absolute": {
+            "default": False,
+            "type": "bool",
+            "help": "whether the path to the stored file is absolute or relative in the output result inc. formatters (Warning: saving an absolute path will show your computer's file structure)",
+        },
    },
    "description": """
    LocalStorage: A storage module for saving archived content locally on the filesystem.
@@ -31,5 +35,5 @@
    ### Notes
    - Default storage folder is `./archived`, but this can be changed via the `save_to` configuration.
    - The `save_absolute` option can reveal the file structure in output formats; use with caution.
-    """
+    """,
 }
--- a/src/auto_archiver/modules/local_storage/local_storage.py
+++ b/src/auto_archiver/modules/local_storage/local_storage.py
@@ -1,4 +1,3 @@
-
 import shutil
 from typing import IO
 import os
@@ -6,25 +5,43 @@ from loguru import logger

 from auto_archiver.core import Media
 from auto_archiver.core import Storage
+from auto_archiver.core.consts import SetupError


 class LocalStorage(Storage):
+    def setup(self) -> None:
+        if len(self.save_to) > 200:
+            raise SetupError(
+                "Your save_to path is too long, this will cause issues saving files on your computer. Please use a shorter path."
+            )

    def get_cdn_url(self, media: Media) -> str:
-        # TODO: is this viable with Storage.configs on path/filename?
-        dest = os.path.join(self.save_to, media.key)
+        dest = media.key
+
        if self.save_absolute:
            dest = os.path.abspath(dest)
        return dest

+    def set_key(self, media, url, metadata):
+        # clarify we want to save the file to the save_to folder
+
+        old_folder = metadata.get("folder", "")
+        metadata.set_context("folder", os.path.join(self.save_to, metadata.get("folder", "")))
+        super().set_key(media, url, metadata)
+        # don't impact other storages that might want a different 'folder' set
+        metadata.set_context("folder", old_folder)
+
    def upload(self, media: Media, **kwargs) -> bool:
        # override parent so that we can use shutil.copy2 and keep metadata
-        dest = os.path.join(self.save_to, media.key)
+        dest = media.key
+
        os.makedirs(os.path.dirname(dest), exist_ok=True)
-        logger.debug(f'[{self.__class__.__name__}] storing file {media.filename} with key {media.key} to {dest}')
+        logger.debug(f"[{self.__class__.__name__}] storing file {media.filename} with key {media.key} to {dest}")
+
        res = shutil.copy2(media.filename, dest)
        logger.info(res)
        return True

    # must be implemented even if unused
-    def uploadf(self, file: IO[bytes], key: str, **kwargs: dict) -> bool: pass
+    def uploadf(self, file: IO[bytes], key: str, **kwargs: dict) -> bool:
+        pass
--- a/src/auto_archiver/modules/meta_enricher/manifest.py
+++ b/src/auto_archiver/modules/meta_enricher/manifest.py
@@ -3,7 +3,7 @@
    "type": ["enricher"],
    "requires_setup": False,
    "dependencies": {
-                          "python": ["loguru"],
+        "python": ["loguru"],
    },
    "description": """ 
    Adds metadata information about the archive operations, Adds metadata about archive operations, including file sizes and archive duration./
--- a/src/auto_archiver/modules/meta_enricher/meta_enricher.py
+++ b/src/auto_archiver/modules/meta_enricher/meta_enricher.py
@@ -23,7 +23,9 @@ class MetaEnricher(Enricher):
        self.enrich_archive_duration(to_enrich)

    def enrich_file_sizes(self, to_enrich: Metadata):
-        logger.debug(f"calculating archive file sizes for url={to_enrich.get_url()} ({len(to_enrich.media)} media files)")
+        logger.debug(
+            f"calculating archive file sizes for url={to_enrich.get_url()} ({len(to_enrich.media)} media files)"
+        )
        total_size = 0
        for media in to_enrich.get_all_media():
            file_stats = os.stat(media.filename)
@@ -34,7 +36,6 @@ class MetaEnricher(Enricher):
        to_enrich.set("total_bytes", total_size)
        to_enrich.set("total_size", self.human_readable_bytes(total_size))

-
    def human_readable_bytes(self, size: int) -> str:
        # receives number of bytes and returns human readble size
        for unit in ["bytes", "KB", "MB", "GB", "TB"]:
@@ -46,4 +47,4 @@ class MetaEnricher(Enricher):
        logger.debug(f"calculating archive duration for url={to_enrich.get_url()} ")

        archive_duration = datetime.datetime.now(datetime.timezone.utc) - to_enrich.get("_processed_at")
-        to_enrich.set("archive_duration_seconds", archive_duration.seconds)
+        to_enrich.set("archive_duration_seconds", archive_duration.seconds)
--- a/src/auto_archiver/modules/metadata_enricher/init.py
+++ b/src/auto_archiver/modules/metadata_enricher/init.py
@@ -1 +1 @@
-from .metadata_enricher import MetadataEnricher
+from .metadata_enricher import MetadataEnricher
--- a/src/auto_archiver/modules/metadata_enricher/manifest.py
+++ b/src/auto_archiver/modules/metadata_enricher/manifest.py
@@ -2,10 +2,7 @@
    "name": "Media Metadata Enricher",
    "type": ["enricher"],
    "requires_setup": True,
-    "dependencies": {
-        "python": ["loguru"], 
-        "bin": ["exiftool"]
-    },
+    "dependencies": {"python": ["loguru"], "bin": ["exiftool"]},
    "description": """
    Extracts metadata information from files using ExifTool.

@@ -17,5 +14,5 @@
    ### Notes
    - Requires ExifTool to be installed and accessible via the system's PATH.
    - Skips enrichment for files where metadata extraction fails.
-    """
+    """,
 }
--- a/src/auto_archiver/modules/metadata_enricher/metadata_enricher.py
+++ b/src/auto_archiver/modules/metadata_enricher/metadata_enricher.py
@@ -11,7 +11,6 @@ class MetadataEnricher(Enricher):
    Extracts metadata information from files using exiftool.
    """

-
    def enrich(self, to_enrich: Metadata) -> None:
        url = to_enrich.get_url()
        logger.debug(f"extracting EXIF metadata for {url=}")
@@ -23,13 +22,13 @@ class MetadataEnricher(Enricher):
    def get_metadata(self, filename: str) -> dict:
        try:
            # Run ExifTool command to extract metadata from the file
-            cmd = ['exiftool', filename]
+            cmd = ["exiftool", filename]
            result = subprocess.run(cmd, capture_output=True, text=True)

            # Process the output to extract individual metadata fields
            metadata = {}
            for line in result.stdout.splitlines():
-                field, value = line.strip().split(':', 1)
+                field, value = line.strip().split(":", 1)
                metadata[field.strip()] = value.strip()
            return metadata
        except FileNotFoundError:
--- a/src/auto_archiver/modules/mute_formatter/manifest.py
+++ b/src/auto_archiver/modules/mute_formatter/manifest.py
@@ -2,8 +2,7 @@
    "name": "Mute Formatter",
    "type": ["formatter"],
    "requires_setup": True,
-    "dependencies": {
-    },
+    "dependencies": {},
    "description": """ Default formatter.
    """,
 }
--- a/src/auto_archiver/modules/mute_formatter/mute_formatter.py
+++ b/src/auto_archiver/modules/mute_formatter/mute_formatter.py
@@ -5,5 +5,5 @@ from auto_archiver.core import Formatter


 class MuteFormatter(Formatter):
-
-    def format(self, item: Metadata) -> Media: return None
+    def format(self, item: Metadata) -> Media:
+        return None
--- a/src/auto_archiver/modules/opentimestamps_enricher/manifest.py
+++ b/src/auto_archiver/modules/opentimestamps_enricher/manifest.py
@@ -0,0 +1,100 @@
+{
+    "name": "OpenTimestamps Enricher",
+    "type": ["enricher"],
+    "requires_setup": True,
+    "dependencies": {
+        "python": [
+            "loguru",
+            "opentimestamps",
+        ],
+    },
+    "configs": {
+        "calendar_urls": {
+            "default": [
+                "https://alice.btc.calendar.opentimestamps.org",
+                "https://bob.btc.calendar.opentimestamps.org",
+                "https://finney.calendar.eternitywall.com",
+                # "https://ots.btc.catallaxy.com/", # ipv4 only
+            ],
+            "help": "List of OpenTimestamps calendar servers to use for timestamping. See here for a list of calendars maintained by opentimestamps:\
+https://opentimestamps.org/#calendars",
+            "type": "list",
+        },
+        "calendar_whitelist": {
+            "default": [],
+            "help": "Optional whitelist of calendar servers. Override this if you are using your own calendar servers. e.g. ['https://mycalendar.com']",
+            "type": "list",
+        },
+    },
+    "description": """
+    Creates OpenTimestamps proofs for archived files, providing blockchain-backed evidence of file existence at a specific time.
+
+    Uses OpenTimestamps – a service that timestamps data using the Bitcoin blockchain, providing a decentralized 
+    and secure way to prove that data existed at a certain point in time. A SHA256 hash of the file to be timestamped is used as the token
+    and sent to each of the 'timestamp calendars' for inclusion in the blockchain. The proof is then saved alongside the original file in a file with
+    the '.ots' extension.
+
+    ### Features
+    - Creates cryptographic timestamp proofs that link files to the Bitcoin
+    - Verifies timestamp proofs have been submitted to the blockchain (note: does not confirm they have been *added*)
+    - Can use multiple calendar servers to ensure reliability and redundancy
+    - Stores timestamp proofs alongside original files for future verification
+
+    ### Timestamp status
+    An opentimestamp, when submitted to a timestmap server will have a 'pending' status (Pending Attestation) as it waits to be added
+    to the blockchain. Once it has been added to the blockchain, it will have a 'confirmed' status (Bitcoin Block Timestamp).
+    This process typically takes several hours, depending on the calendar server and the current state of the Bitcoin network. As such,
+    the status of all timestamps added will be 'pending' until they are subsequently confirmed (see 'Upgrading Timestamps' below).
+
+    There are two possible statuses for a timestamp:
+    - `Pending`: The timestamp has been submitted to the calendar server but has not yet been confirmed in the Bitcoin blockchain.
+    - `Confirmed`: The timestamp has been confirmed in the Bitcoin blockchain.
+
+    ### Upgrading Timestamps
+    To upgrade a timestamp from 'pending' to 'confirmed', you can use the `ots upgrade` command from the opentimestamps-client package
+    (install it with `pip install opentimesptamps-client`).
+    Example: `ots upgrade my_file.ots`
+
+    Here is a useful script that could be used to upgrade all timestamps in a directory, which could be run on a cron job:
+```{code}  bash
+find . -name "*.ots" -type f | while read file; do
+    echo "Upgrading OTS $file"
+    ots upgrade $file
+done
+# The result might look like:
+# Upgrading OTS ./my_file.ots
+# Got 1 attestation(s) from https://alice.btc.calendar.opentimestamps.org
+# Success! Timestamp complete
+```
+
+```{note} Note: this will only upgrade the .ots files, and will not change the status text in any output .html files or any databases where the
+metadata is stored (e.g. Google Sheets, CSV database, API database etc.).
+```
+
+    ### Verifying Timestamps
+    The easiest way to verify a timestamp (ots) file is to install the opentimestamps-client command line tool and use the `ots verify` command.
+    Example: `ots verify my_file.ots`
+
+    ```{code}  bash
+$ ots verify my_file.ots
+Calendar https://bob.btc.calendar.opentimestamps.org: Pending confirmation in Bitcoin blockchain
+Calendar https://finney.calendar.eternitywall.com: Pending confirmation in Bitcoin blockchain
+Calendar https://alice.btc.calendar.opentimestamps.org: Timestamped by transaction 12345; waiting for 6 confirmations
+```
+
+    Note: if you're using a storage with `filename_generator` set to `static` or `random`, the files will be renamed when they are saved to the
+    final location meaning you will need to specify the original filename when verifying the timestamp with `ots verify -f original_filename my_file.ots`.
+
+    ### Choosing Calendar Servers
+
+    By default, the OpenTimestamps enricher uses a set of public calendar servers provided by the 'opentimestamps' project.
+    You can customize the list of calendar servers by providing URLs in the `calendar_urls` configuration option.
+
+    ### Calendar WhiteList
+
+    By default, the opentimestamps package only allows their own calendars to be used (see `DEFAULT_CALENDAR_WHITELIST` in `opentimestamps.calendar`),
+    if you want to use your own calendars, then you can override this setting in the `calendar_whitelist` configuration option.
+
+   
+    """,
+}
--- a/src/auto_archiver/modules/opentimestamps_enricher/opentimestamps_enricher.py
+++ b/src/auto_archiver/modules/opentimestamps_enricher/opentimestamps_enricher.py
@@ -0,0 +1,172 @@
+import os
+
+from loguru import logger
+import opentimestamps
+from opentimestamps.calendar import RemoteCalendar, DEFAULT_CALENDAR_WHITELIST
+from opentimestamps.core.timestamp import Timestamp, DetachedTimestampFile
+from opentimestamps.core.notary import PendingAttestation, BitcoinBlockHeaderAttestation
+from opentimestamps.core.op import OpSHA256
+from opentimestamps.core import serialize
+from auto_archiver.core import Enricher
+from auto_archiver.core import Metadata, Media
+from auto_archiver.utils.misc import get_current_timestamp
+
+
+class OpentimestampsEnricher(Enricher):
+    def enrich(self, to_enrich: Metadata) -> None:
+        url = to_enrich.get_url()
+        logger.debug(f"OpenTimestamps timestamping files for {url=}")
+
+        # Get the media files to timestamp
+        media_files = [m for m in to_enrich.media if m.filename and not m.get("opentimestamps")]
+        if not media_files:
+            logger.warning(f"No files found to timestamp in {url=}")
+            return
+
+        timestamp_files = []
+        for media in media_files:
+            try:
+                # Get the file path from the media
+                file_path = media.filename
+                if not os.path.exists(file_path):
+                    logger.warning(f"File not found: {file_path}")
+                    continue
+
+                # Create timestamp for the file - hash is SHA256
+                # Note: hash is hard-coded to SHA256 and does not use hash_enricher to set it.
+                # SHA256 is the recommended hash, ref: https://github.com/bellingcat/auto-archiver/pull/247#discussion_r1992433181
+                logger.debug(f"Creating timestamp for {file_path}")
+                file_hash = None
+                with open(file_path, "rb") as f:
+                    file_hash = OpSHA256().hash_fd(f)
+
+                if not file_hash:
+                    logger.warning(f"Failed to hash file for timestamping, skipping: {file_path}")
+                    continue
+
+                # Create a timestamp with the file hash
+                timestamp = Timestamp(file_hash)
+
+                # Create a detached timestamp file with the hash operation and timestamp
+                detached_timestamp = DetachedTimestampFile(OpSHA256(), timestamp)
+
+                # Submit to calendar servers
+                submitted_to_calendar = False
+
+                logger.debug(f"Submitting timestamp to calendar servers for {file_path}")
+                calendars = []
+                whitelist = DEFAULT_CALENDAR_WHITELIST
+
+                if self.calendar_whitelist:
+                    whitelist = set(self.calendar_whitelist)
+
+                # Create calendar instances
+                calendar_urls = []
+                for url in self.calendar_urls:
+                    if url in whitelist:
+                        calendars.append(RemoteCalendar(url))
+                        calendar_urls.append(url)
+
+                # Submit the hash to each calendar
+                for calendar in calendars:
+                    try:
+                        calendar_timestamp = calendar.submit(file_hash)
+                        timestamp.merge(calendar_timestamp)
+                        logger.debug(f"Successfully submitted to calendar: {calendar.url}")
+                        submitted_to_calendar = True
+                    except Exception as e:
+                        logger.warning(f"Failed to submit to calendar {calendar.url}: {e}")
+
+                # If all calendar submissions failed, add pending attestations
+                if not submitted_to_calendar and not timestamp.attestations:
+                    logger.error(
+                        f"Failed to submit to any calendar for {file_path}. **This file will not be timestamped.**"
+                    )
+                    media.set("opentimestamps", False)
+                    continue
+
+                # Save the timestamp proof to a file
+                timestamp_path = os.path.join(self.tmp_dir, f"{os.path.basename(file_path)}.ots")
+                try:
+                    with open(timestamp_path, "wb") as f:
+                        # Create a serialization context and write to the file
+                        ctx = serialize.BytesSerializationContext()
+                        detached_timestamp.serialize(ctx)
+                        f.write(ctx.getbytes())
+                except Exception as e:
+                    logger.warning(f"Failed to serialize timestamp file: {e}")
+                    continue
+
+                # Create media for the timestamp file
+                timestamp_media = Media(filename=timestamp_path)
+                # explicitly set the mimetype, normally .ots files are 'application/vnd.oasis.opendocument.spreadsheet-template'
+                timestamp_media.mimetype = "application/vnd.opentimestamps"
+                timestamp_media.set("opentimestamps_version", opentimestamps.__version__)
+
+                verification_info = self.verify_timestamp(detached_timestamp)
+                for key, value in verification_info.items():
+                    timestamp_media.set(key, value)
+
+                media.set("opentimestamp_files", [timestamp_media])
+                timestamp_files.append(timestamp_media.filename)
+                # Update the original media to indicate it's been timestamped
+                media.set("opentimestamps", True)
+
+            except Exception as e:
+                logger.warning(f"Error while timestamping {media.filename}: {e}")
+
+        # Add timestamp files to the metadata
+        if timestamp_files:
+            to_enrich.set("opentimestamped", True)
+            to_enrich.set("opentimestamps_count", len(timestamp_files))
+            logger.success(f"{len(timestamp_files)} OpenTimestamps proofs created for {url=}")
+        else:
+            to_enrich.set("opentimestamped", False)
+            logger.warning(f"No successful timestamps created for {url=}")
+
+    def verify_timestamp(self, detached_timestamp):
+        """
+        Verify a timestamp and extract verification information.
+
+        Args:
+            detached_timestamp: The detached timestamp to verify.
+
+        Returns:
+            dict: Information about the verification result.
+        """
+        result = {}
+
+        # Check if we have attestations
+        attestations = list(detached_timestamp.timestamp.all_attestations())
+        result["attestation_count"] = len(attestations)
+
+        if attestations:
+            attestation_info = []
+            for msg, attestation in attestations:
+                info = {}
+
+                # Process different types of attestations
+                if isinstance(attestation, PendingAttestation):
+                    info["status"] = "pending"
+                    info["uri"] = attestation.uri
+
+                elif isinstance(attestation, BitcoinBlockHeaderAttestation):
+                    info["status"] = "confirmed"
+                    info["block_height"] = attestation.height
+
+                info["last_check"] = get_current_timestamp()
+
+                attestation_info.append(info)
+
+            result["attestations"] = attestation_info
+
+            # For at least one confirmed attestation
+            if any("confirmed" in a.get("status") for a in attestation_info):
+                result["verified"] = True
+            else:
+                result["verified"] = False
+        else:
+            result["verified"] = False
+        result["last_updated"] = get_current_timestamp()
+
+        return result
--- a/src/auto_archiver/modules/pdq_hash_enricher/init.py
+++ b/src/auto_archiver/modules/pdq_hash_enricher/init.py
@@ -1 +1 @@
-from .pdq_hash_enricher import PdqHashEnricher
+from .pdq_hash_enricher import PdqHashEnricher
--- a/src/auto_archiver/modules/pdq_hash_enricher/manifest.py
+++ b/src/auto_archiver/modules/pdq_hash_enricher/manifest.py
@@ -17,5 +17,5 @@
    ### Notes
    - Best used after enrichers like `thumbnail_enricher` or `screenshot_enricher` to ensure images are available.
    - Uses the `pdqhash` library to compute 256-bit perceptual hashes, which are stored as hexadecimal strings.
-    """
+    """,
 }
--- a/src/auto_archiver/modules/pdq_hash_enricher/pdq_hash_enricher.py
+++ b/src/auto_archiver/modules/pdq_hash_enricher/pdq_hash_enricher.py
@@ -10,6 +10,7 @@ This enricher is typically used after thumbnail or screenshot enrichers
 to ensure images are available for hashing.

 """
+
 import traceback
 import pdqhash
 import numpy as np
@@ -34,7 +35,12 @@ class PdqHashEnricher(Enricher):
        for m in to_enrich.media:
            for media in m.all_inner_media(True):
                media_id = media.get("id", "")
-                if media.is_image() and "screenshot" not in media_id and "warc-file-" not in media_id and len(hd := self.calculate_pdq_hash(media.filename)):
+                if (
+                    media.is_image()
+                    and "screenshot" not in media_id
+                    and "warc-file-" not in media_id
+                    and len(hd := self.calculate_pdq_hash(media.filename))
+                ):
                    media.set("pdq_hash", hd)
                    media_with_hashes.append(media.filename)

@@ -51,5 +57,7 @@ class PdqHashEnricher(Enricher):
                hash = "".join(str(b) for b in hash_array)
                return hex(int(hash, 2))[2:]
        except UnidentifiedImageError as e:
-            logger.error(f"Image {filename=} is likely corrupted or in unsupported format {e}: {traceback.format_exc()}")
+            logger.error(
+                f"Image {filename=} is likely corrupted or in unsupported format {e}: {traceback.format_exc()}"
+            )
        return ""
--- a/src/auto_archiver/modules/s3_storage/init.py
+++ b/src/auto_archiver/modules/s3_storage/init.py
@@ -1 +1 @@
-from .s3_storage import S3Storage
+from .s3_storage import S3Storage
--- a/src/auto_archiver/modules/s3_storage/manifest.py
+++ b/src/auto_archiver/modules/s3_storage/manifest.py
@@ -13,27 +13,27 @@
        },
        "filename_generator": {
            "default": "static",
-            "help": "how to name stored files: 'random' creates a random string; 'static' uses a replicable strategy such as a hash.",
+            "help": "how to name stored files: 'random' creates a random string; 'static' uses a hash, with the settings of the 'hash_enricher' module (defaults to SHA256 if not enabled).",
            "choices": ["random", "static"],
        },
        "bucket": {"default": None, "help": "S3 bucket name"},
        "region": {"default": None, "help": "S3 region name"},
        "key": {"default": None, "help": "S3 API key"},
        "secret": {"default": None, "help": "S3 API secret"},
-        "random_no_duplicate": {"default": False,
-                                "type": "bool",
-                                "help": "if set, it will override `path_generator`, `filename_generator` and `folder`. It will check if the file already exists and if so it will not upload it again. Creates a new root folder path `no-dups/`"},
+        "random_no_duplicate": {
+            "default": False,
+            "type": "bool",
+            "help": "if set, it will override `path_generator`, `filename_generator` and `folder`. It will check if the file already exists and if so it will not upload it again. Creates a new root folder path `no-dups/`",
+        },
        "endpoint_url": {
-            "default": 'https://{region}.digitaloceanspaces.com',
-            "help": "S3 bucket endpoint, {region} are inserted at runtime"
+            "default": "https://{region}.digitaloceanspaces.com",
+            "help": "S3 bucket endpoint, {region} are inserted at runtime",
        },
        "cdn_url": {
-            "default": 'https://{bucket}.{region}.cdn.digitaloceanspaces.com/{key}',
-            "help": "S3 CDN url, {bucket}, {region} and {key} are inserted at runtime"
+            "default": "https://{bucket}.{region}.cdn.digitaloceanspaces.com/{key}",
+            "help": "S3 CDN url, {bucket}, {region} and {key} are inserted at runtime",
        },
-        "private": {"default": False,
-                    "type": "bool",
-                    "help": "if true S3 files will not be readable online"},
+        "private": {"default": False, "type": "bool", "help": "if true S3 files will not be readable online"},
    },
    "description": """
    S3Storage: A storage module for saving media files to an S3-compatible object storage.
@@ -50,5 +50,5 @@
    - The `random_no_duplicate` option ensures no duplicate uploads by leveraging hash-based folder structures.
    - Uses `boto3` for interaction with the S3 API.
    - Depends on the `HashEnricher` module for hash calculation.
-    """
+    """,
 }
--- a/src/auto_archiver/modules/s3_storage/s3_storage.py
+++ b/src/auto_archiver/modules/s3_storage/s3_storage.py
@@ -1,4 +1,3 @@
-
 from typing import IO

 import boto3
@@ -11,60 +10,62 @@ from auto_archiver.utils.misc import calculate_file_hash, random_str

 NO_DUPLICATES_FOLDER = "no-dups/"

-class S3Storage(Storage):

+class S3Storage(Storage):
    def setup(self) -> None:
        self.s3 = boto3.client(
-            's3',
+            "s3",
            region_name=self.region,
            endpoint_url=self.endpoint_url.format(region=self.region),
            aws_access_key_id=self.key,
-            aws_secret_access_key=self.secret
+            aws_secret_access_key=self.secret,
        )
        if self.random_no_duplicate:
-            logger.warning("random_no_duplicate is set to True, this will override `path_generator`, `filename_generator` and `folder`.")
+            logger.warning(
+                "random_no_duplicate is set to True, this will override `path_generator`, `filename_generator` and `folder`."
+            )

    def get_cdn_url(self, media: Media) -> str:
        return self.cdn_url.format(bucket=self.bucket, region=self.region, key=media.key)

    def uploadf(self, file: IO[bytes], media: Media, **kwargs: dict) -> None:
-        if not self.is_upload_needed(media): return True
+        if not self.is_upload_needed(media):
+            return True

        extra_args = kwargs.get("extra_args", {})
-        if not self.private and 'ACL' not in extra_args:
-            extra_args['ACL'] = 'public-read'
+        if not self.private and "ACL" not in extra_args:
+            extra_args["ACL"] = "public-read"

-        if 'ContentType' not in extra_args:
+        if "ContentType" not in extra_args:
            try:
                if media.mimetype:
-                    extra_args['ContentType'] = media.mimetype
+                    extra_args["ContentType"] = media.mimetype
            except Exception as e:
                logger.warning(f"Unable to get mimetype for {media.key=}, error: {e}")
        self.s3.upload_fileobj(file, Bucket=self.bucket, Key=media.key, ExtraArgs=extra_args)
        return True
-    
+
    def is_upload_needed(self, media: Media) -> bool:
        if self.random_no_duplicate:
            # checks if a folder with the hash already exists, if so it skips the upload
            hd = calculate_file_hash(media.filename)
            path = os.path.join(NO_DUPLICATES_FOLDER, hd[:24])

-            if existing_key:=self.file_in_folder(path):
-                media.key = existing_key
+            if existing_key := self.file_in_folder(path):
+                media._key = existing_key
                media.set("previously archived", True)
                logger.debug(f"skipping upload of {media.filename} because it already exists in {media.key}")
                return False
-            
+
            _, ext = os.path.splitext(media.key)
-            media.key = os.path.join(path, f"{random_str(24)}{ext}")
+            media._key = os.path.join(path, f"{random_str(24)}{ext}")
        return True

-    def file_in_folder(self, path:str) -> str:
+    def file_in_folder(self, path: str) -> str:
        # checks if path exists and is not an empty folder
-        if not path.endswith('/'):
-            path = path + '/' 
-        resp = self.s3.list_objects(Bucket=self.bucket, Prefix=path, Delimiter='/', MaxKeys=1)
-        if 'Contents' in resp:
-            return resp['Contents'][0]['Key']
+        if not path.endswith("/"):
+            path = path + "/"
+        resp = self.s3.list_objects(Bucket=self.bucket, Prefix=path, Delimiter="/", MaxKeys=1)
+        if "Contents" in resp:
+            return resp["Contents"][0]["Key"]
        return False
-
--- a/src/auto_archiver/modules/screenshot_enricher/manifest.py
+++ b/src/auto_archiver/modules/screenshot_enricher/manifest.py
@@ -6,14 +6,29 @@
        "python": ["loguru", "selenium"],
    },
    "configs": {
-            "width": {"default": 1280, "help": "width of the screenshots"},
-            "height": {"default": 720, "help": "height of the screenshots"},
-            "timeout": {"default": 60, "help": "timeout for taking the screenshot"},
-            "sleep_before_screenshot": {"default": 4, "help": "seconds to wait for the pages to load before taking screenshot"},
-            "http_proxy": {"default": "", "help": "http proxy to use for the webdriver, eg http://proxy-user:password@proxy-ip:port"},
-            "save_to_pdf": {"default": False, "help": "save the page as pdf along with the screenshot. PDF saving options can be adjusted with the 'print_options' parameter"},
-            "print_options": {"default": {}, "help": "options to pass to the pdf printer"}
+        "width": {"default": 1280, "type": "int", "help": "width of the screenshots"},
+        "height": {"default": 1024, "type": "int", "help": "height of the screenshots"},
+        "timeout": {"default": 60, "type": "int", "help": "timeout for taking the screenshot"},
+        "sleep_before_screenshot": {
+            "default": 4,
+            "type": "int",
+            "help": "seconds to wait for the pages to load before taking screenshot",
        },
+        "http_proxy": {
+            "default": "",
+            "help": "http proxy to use for the webdriver, eg http://proxy-user:password@proxy-ip:port",
+        },
+        "save_to_pdf": {
+            "default": False,
+            "type": "bool",
+            "help": "save the page as pdf along with the screenshot. PDF saving options can be adjusted with the 'print_options' parameter",
+        },
+        "print_options": {
+            "default": {},
+            "help": "options to pass to the pdf printer, in JSON format. See https://www.selenium.dev/documentation/webdriver/interactions/print_page/ for more information",
+            "type": "json_loader",
+        },
+    },
    "description": """
    Captures screenshots and optionally saves web pages as PDFs using a WebDriver.

@@ -25,5 +40,5 @@

    ### Notes
    - Requires a WebDriver (e.g., ChromeDriver) installed and accessible via the system's PATH.
-    """
+    """,
 }
--- a/src/auto_archiver/modules/screenshot_enricher/screenshot_enricher.py
+++ b/src/auto_archiver/modules/screenshot_enricher/screenshot_enricher.py
@@ -1,5 +1,6 @@
 from loguru import logger
-import time, os
+import time
+import os
 import base64

 from selenium.common.exceptions import TimeoutException
@@ -9,8 +10,8 @@ from auto_archiver.core import Enricher
 from auto_archiver.utils import Webdriver, url as UrlUtil, random_str
 from auto_archiver.core import Media, Metadata

-class ScreenshotEnricher(Enricher):

+class ScreenshotEnricher(Enricher):
    def __init__(self, webdriver_factory=None):
        super().__init__()
        self.webdriver_factory = webdriver_factory or Webdriver
@@ -25,8 +26,14 @@ class ScreenshotEnricher(Enricher):
        logger.debug(f"Enriching screenshot for {url=}")
        auth = self.auth_for_site(url)
        with self.webdriver_factory(
-                self.width, self.height, self.timeout, facebook_accept_cookies='facebook.com' in url,
-                       http_proxy=self.http_proxy, print_options=self.print_options, auth=auth) as driver:
+            self.width,
+            self.height,
+            self.timeout,
+            facebook_accept_cookies="facebook.com" in url,
+            http_proxy=self.http_proxy,
+            print_options=self.print_options,
+            auth=auth,
+        ) as driver:
            try:
                driver.get(url)
                time.sleep(int(self.sleep_before_screenshot))
@@ -43,4 +50,3 @@ class ScreenshotEnricher(Enricher):
                logger.info("TimeoutException loading page for screenshot")
            except Exception as e:
                logger.error(f"Got error while loading webdriver for screenshot enricher: {e}")
-
--- a/src/auto_archiver/modules/ssl_enricher/init.py
+++ b/src/auto_archiver/modules/ssl_enricher/init.py
@@ -1 +1 @@
-from .ssl_enricher import SSLEnricher
+from .ssl_enricher import SSLEnricher
--- a/src/auto_archiver/modules/ssl_enricher/manifest.py
+++ b/src/auto_archiver/modules/ssl_enricher/manifest.py
@@ -5,9 +5,13 @@
    "dependencies": {
        "python": ["loguru", "slugify"],
    },
-    'entry_point': 'ssl_enricher::SSLEnricher',
+    "entry_point": "ssl_enricher::SSLEnricher",
    "configs": {
-        "skip_when_nothing_archived": {"default": True, "help": "if true, will skip enriching when no media is archived"},
+        "skip_when_nothing_archived": {
+            "default": True,
+            "type": "bool",
+            "help": "if true, will skip enriching when no media is archived",
+        },
    },
    "description": """
    Retrieves SSL certificate information for a domain and stores it as a file.
@@ -19,5 +23,5 @@

    ### Notes
    - Requires the target URL to use the HTTPS scheme; other schemes are not supported.
-    """
+    """,
 }
--- a/src/auto_archiver/modules/ssl_enricher/ssl_enricher.py
+++ b/src/auto_archiver/modules/ssl_enricher/ssl_enricher.py
@@ -1,4 +1,5 @@
-import ssl, os
+import ssl
+import os
 from slugify import slugify
 from urllib.parse import urlparse
 from loguru import logger
@@ -13,16 +14,18 @@ class SSLEnricher(Enricher):
    """

    def enrich(self, to_enrich: Metadata) -> None:
-        if not to_enrich.media and self.skip_when_nothing_archived: return
-        
+        if not to_enrich.media and self.skip_when_nothing_archived:
+            return
+
        url = to_enrich.get_url()
        parsed = urlparse(url)
        assert parsed.scheme in ["https"], f"Invalid URL scheme {url=}"
-        
+
        domain = parsed.netloc
        logger.debug(f"fetching SSL certificate for {domain=} in {url=}")

        cert = ssl.get_server_certificate((domain, 443))
        cert_fn = os.path.join(self.tmp_dir, f"{slugify(domain)}.pem")
-        with open(cert_fn, "w") as f: f.write(cert)
+        with open(cert_fn, "w") as f:
+            f.write(cert)
        to_enrich.add_media(Media(filename=cert_fn), id="ssl_certificate")
--- a/src/auto_archiver/modules/telegram_extractor/init.py
+++ b/src/auto_archiver/modules/telegram_extractor/init.py
@@ -1 +1 @@
-from .telegram_extractor import TelegramExtractor
+from .telegram_extractor import TelegramExtractor
--- a/src/auto_archiver/modules/telegram_extractor/telegram_extractor.py
+++ b/src/auto_archiver/modules/telegram_extractor/telegram_extractor.py
@@ -1,4 +1,6 @@
-import requests, re, html
+import requests
+import re
+import html
 from bs4 import BeautifulSoup
 from loguru import logger

@@ -15,11 +17,11 @@ class TelegramExtractor(Extractor):
    def download(self, item: Metadata) -> Metadata:
        url = item.get_url()
        # detect URLs that we definitely cannot handle
-        if 't.me' != item.netloc:
+        if "t.me" != item.netloc:
            return False

        headers = {
-            'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/81.0.4044.138 Safari/537.36'
+            "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/81.0.4044.138 Safari/537.36"
        }

        # TODO: check if we can do this more resilient to variable URLs
@@ -27,11 +29,11 @@ class TelegramExtractor(Extractor):
            url += "?embed=1"

        t = requests.get(url, headers=headers)
-        s = BeautifulSoup(t.content, 'html.parser')
+        s = BeautifulSoup(t.content, "html.parser")

        result = Metadata()
        result.set_content(html.escape(str(t.content)))
-        if (timestamp := (s.find_all('time') or [{}])[0].get('datetime')):
+        if timestamp := (s.find_all("time") or [{}])[0].get("datetime"):
            result.set_timestamp(timestamp)

        video = s.find("video")
@@ -41,25 +43,26 @@ class TelegramExtractor(Extractor):

            image_urls = []
            for im in image_tags:
-                urls = [u.replace("'", "") for u in re.findall(r'url\((.*?)\)', im['style'])]
+                urls = [u.replace("'", "") for u in re.findall(r"url\((.*?)\)", im["style"])]
                image_urls += urls

-            if not len(image_urls): return False
+            if not len(image_urls):
+                return False
            for img_url in image_urls:
                result.add_media(Media(self.download_from_url(img_url)))
        else:
-            video_url = video.get('src')
+            video_url = video.get("src")
            m_video = Media(self.download_from_url(video_url))
            # extract duration from HTML
            try:
-                duration = s.find_all('time')[0].contents[0]
-                if ':' in duration:
-                    duration = float(duration.split(
-                        ':')[0]) * 60 + float(duration.split(':')[1])
+                duration = s.find_all("time")[0].contents[0]
+                if ":" in duration:
+                    duration = float(duration.split(":")[0]) * 60 + float(duration.split(":")[1])
                else:
                    duration = float(duration)
                m_video.set("duration", duration)
-            except: pass
+            except Exception:
+                pass
            result.add_media(m_video)

        return result.success("telegram")
--- a/src/auto_archiver/modules/telethon_extractor/init.py
+++ b/src/auto_archiver/modules/telethon_extractor/init.py
@@ -1 +1 @@
-from .telethon_extractor import TelethonExtractor
+from .telethon_extractor import TelethonExtractor
--- a/src/auto_archiver/modules/telethon_extractor/manifest.py
+++ b/src/auto_archiver/modules/telethon_extractor/manifest.py
@@ -3,24 +3,35 @@
    "type": ["extractor"],
    "requires_setup": True,
    "dependencies": {
-        "python": ["telethon",
-                   "loguru",
-                   "tqdm",
-                   ],
-        "bin": [""]
+        "python": [
+            "telethon",
+            "loguru",
+            "tqdm",
+        ],
+        "bin": [""],
    },
    "configs": {
-            "api_id": {"default": None, "help": "telegram API_ID value, go to https://my.telegram.org/apps"},
-            "api_hash": {"default": None, "help": "telegram API_HASH value, go to https://my.telegram.org/apps"},
-            "bot_token": {"default": None, "help": "optional, but allows access to more content such as large videos, talk to @botfather"},
-            "session_file": {"default": "secrets/anon", "help": "optional, records the telegram login session for future usage, '.session' will be appended to the provided value."},
-            "join_channels": {"default": True, "help": "disables the initial setup with channel_invites config, useful if you have a lot and get stuck"},
-            "channel_invites": {
-                "default": {},
-                "help": "(JSON string) private channel invite links (format: t.me/joinchat/HASH OR t.me/+HASH) and (optional but important to avoid hanging for minutes on startup) channel id (format: CHANNEL_ID taken from a post url like https://t.me/c/CHANNEL_ID/1), the telegram account will join any new channels on setup",
-                "type": "auto_archiver.utils.json_loader",
-            }
+        "api_id": {"default": None, "help": "telegram API_ID value, go to https://my.telegram.org/apps"},
+        "api_hash": {"default": None, "help": "telegram API_HASH value, go to https://my.telegram.org/apps"},
+        "bot_token": {
+            "default": None,
+            "help": "optional, but allows access to more content such as large videos, talk to @botfather",
        },
+        "session_file": {
+            "default": "secrets/anon",
+            "help": "optional, records the telegram login session for future usage, '.session' will be appended to the provided value.",
+        },
+        "join_channels": {
+            "default": True,
+            "type": "bool",
+            "help": "disables the initial setup with channel_invites config, useful if you have a lot and get stuck",
+        },
+        "channel_invites": {
+            "default": {},
+            "help": "(JSON string) private channel invite links (format: t.me/joinchat/HASH OR t.me/+HASH) and (optional but important to avoid hanging for minutes on startup) channel id (format: CHANNEL_ID taken from a post url like https://t.me/c/CHANNEL_ID/1), the telegram account will join any new channels on setup",
+            "type": "json_loader",
+        },
+    },
    "description": """
 The `TelethonExtractor` uses the Telethon library to archive posts and media from Telegram channels and groups. 
 It supports private and public channels, downloading grouped posts with media, and can join channels using invite links 
@@ -44,5 +55,5 @@ To use the `TelethonExtractor`, you must configure the following:
 The first time you run, you will be prompted to do a authentication with the phone number associated, alternatively you can put your `anon.session` in the root.


-"""
-}
+""",
+}
--- a/src/auto_archiver/modules/telethon_extractor/telethon_extractor.py
+++ b/src/auto_archiver/modules/telethon_extractor/telethon_extractor.py
@@ -1,12 +1,18 @@
-
 import shutil
 from telethon.sync import TelegramClient
 from telethon.errors import ChannelInvalidError
 from telethon.tl.functions.messages import ImportChatInviteRequest
-from telethon.errors.rpcerrorlist import UserAlreadyParticipantError, FloodWaitError, InviteRequestSentError, InviteHashExpiredError
+from telethon.errors.rpcerrorlist import (
+    UserAlreadyParticipantError,
+    FloodWaitError,
+    InviteRequestSentError,
+    InviteHashExpiredError,
+)
 from loguru import logger
 from tqdm import tqdm
-import re, time, os
+import re
+import time
+import os

 from auto_archiver.core import Extractor
 from auto_archiver.core import Metadata, Media
@@ -17,9 +23,7 @@ class TelethonExtractor(Extractor):
    valid_url = re.compile(r"https:\/\/t\.me(\/c){0,1}\/(.+)\/(\d+)")
    invite_pattern = re.compile(r"t.me(\/joinchat){0,1}\/\+?(.+)")

-
    def setup(self) -> None:
-
        """
        1. makes a copy of session_file that is removed in cleanup
        2. trigger login process for telegram or proceed if already saved in a session file
@@ -34,7 +38,7 @@ class TelethonExtractor(Extractor):

        # initiate the client
        self.client = TelegramClient(self.session_file, self.api_id, self.api_hash)
-        
+
        with self.client.start():
            logger.success(f"SETUP {self.name} login works.")

@@ -52,18 +56,20 @@ class TelethonExtractor(Extractor):
                    channel_invite = self.channel_invites[i]
                    channel_id = channel_invite.get("id", False)
                    invite = channel_invite["invite"]
-                    if (match := self.invite_pattern.search(invite)):
+                    if match := self.invite_pattern.search(invite):
                        try:
                            if channel_id:
                                ent = self.client.get_entity(int(channel_id))  # fails if not a member
                            else:
                                ent = self.client.get_entity(invite)  # fails if not a member
-                                logger.warning(f"please add the property id='{ent.id}' to the 'channel_invites' configuration where {invite=}, not doing so can lead to a minutes-long setup time due to telegram's rate limiting.")
-                        except ValueError as e:
+                                logger.warning(
+                                    f"please add the property id='{ent.id}' to the 'channel_invites' configuration where {invite=}, not doing so can lead to a minutes-long setup time due to telegram's rate limiting."
+                                )
+                        except ValueError:
                            logger.info(f"joining new channel {invite=}")
                            try:
                                self.client(ImportChatInviteRequest(match.group(2)))
-                            except UserAlreadyParticipantError as e:
+                            except UserAlreadyParticipantError:
                                logger.info(f"already joined {invite=}")
                            except InviteRequestSentError:
                                logger.warning(f"already sent a join request with {invite} still no answer")
@@ -95,7 +101,8 @@ class TelethonExtractor(Extractor):
        # detect URLs that we definitely cannot handle
        match = self.valid_url.search(url)
        logger.debug(f"TELETHON: {match=}")
-        if not match: return False
+        if not match:
+            return False

        is_private = match.group(1) == "/c"
        chat = int(match.group(2)) if is_private else match.group(2)
@@ -105,45 +112,53 @@ class TelethonExtractor(Extractor):

        # NB: not using bot_token since then private channels cannot be archived: self.client.start(bot_token=self.bot_token)
        with self.client.start():
-        # with self.client.start(bot_token=self.bot_token):
+            # with self.client.start(bot_token=self.bot_token):
            try:
                post = self.client.get_messages(chat, ids=post_id)
            except ValueError as e:
                logger.error(f"Could not fetch telegram {url} possibly it's private: {e}")
                return False
            except ChannelInvalidError as e:
-                logger.error(f"Could not fetch telegram {url}. This error may be fixed if you setup a bot_token in addition to api_id and api_hash (but then private channels will not be archived, we need to update this logic to handle both): {e}")
+                logger.error(
+                    f"Could not fetch telegram {url}. This error may be fixed if you setup a bot_token in addition to api_id and api_hash (but then private channels will not be archived, we need to update this logic to handle both): {e}"
+                )
                return False

            logger.debug(f"TELETHON GOT POST {post=}")
-            if post is None: return False
+            if post is None:
+                return False

            media_posts = self._get_media_posts_in_group(chat, post)
-            logger.debug(f'got {len(media_posts)=} for {url=}')
+            logger.debug(f"got {len(media_posts)=} for {url=}")

            tmp_dir = self.tmp_dir

            group_id = post.grouped_id if post.grouped_id is not None else post.id
            title = post.message
            for mp in media_posts:
-                if len(mp.message) > len(title): title = mp.message  # save the longest text found (usually only 1)
+                if len(mp.message) > len(title):
+                    title = mp.message  # save the longest text found (usually only 1)

                # media can also be in entities
                if mp.entities:
-                    other_media_urls = [e.url for e in mp.entities if hasattr(e, "url") and e.url and self._guess_file_type(e.url) in ["video", "image", "audio"]]
+                    other_media_urls = [
+                        e.url
+                        for e in mp.entities
+                        if hasattr(e, "url") and e.url and self._guess_file_type(e.url) in ["video", "image", "audio"]
+                    ]
                    if len(other_media_urls):
                        logger.debug(f"Got {len(other_media_urls)} other media urls from {mp.id=}: {other_media_urls}")
                    for i, om_url in enumerate(other_media_urls):
-                        filename = self.download_from_url(om_url, f'{chat}_{group_id}_{i}')
+                        filename = self.download_from_url(om_url, f"{chat}_{group_id}_{i}")
                        result.add_media(Media(filename=filename), id=f"{group_id}_{i}")

-                filename_dest = os.path.join(tmp_dir, f'{chat}_{group_id}', str(mp.id))
+                filename_dest = os.path.join(tmp_dir, f"{chat}_{group_id}", str(mp.id))
                filename = self.client.download_media(mp.media, filename_dest)
                if not filename:
                    logger.debug(f"Empty media found, skipping {str(mp)=}")
                    continue
                result.add_media(Media(filename))
-            
+
            result.set_title(title).set_timestamp(post.date).set("api_data", post.to_dict())
            if post.message != title:
                result.set_content(post.message)
--- a/src/auto_archiver/modules/thumbnail_enricher/manifest.py
+++ b/src/auto_archiver/modules/thumbnail_enricher/manifest.py
@@ -2,18 +2,19 @@
    "name": "Thumbnail Enricher",
    "type": ["enricher"],
    "requires_setup": False,
-    "dependencies": {
-        "python": ["loguru", "ffmpeg"],
-        "bin": ["ffmpeg"]
-    },
+    "dependencies": {"python": ["loguru", "ffmpeg"], "bin": ["ffmpeg"]},
    "configs": {
-            "thumbnails_per_minute": {"default": 60,
-                                      "type": "int",
-                                      "help": "how many thumbnails to generate per minute of video, can be limited by max_thumbnails"},
-            "max_thumbnails": {"default": 16,
-                               "type": "int",
-                               "help": "limit the number of thumbnails to generate per video, 0 means no limit"},
+        "thumbnails_per_minute": {
+            "default": 60,
+            "type": "int",
+            "help": "how many thumbnails to generate per minute of video, can be limited by max_thumbnails",
        },
+        "max_thumbnails": {
+            "default": 16,
+            "type": "int",
+            "help": "limit the number of thumbnails to generate per video, 0 means no limit",
+        },
+    },
    "description": """
    Generates thumbnails for video files to provide visual previews.

@@ -27,5 +28,5 @@
    - Requires `ffmpeg` to be installed and accessible via the system's PATH.
    - Handles videos without pre-existing duration metadata by probing with `ffmpeg`.
    - Skips enrichment for non-video media files.
-    """
+    """,
 }
--- a/src/auto_archiver/modules/thumbnail_enricher/thumbnail_enricher.py
+++ b/src/auto_archiver/modules/thumbnail_enricher/thumbnail_enricher.py
@@ -6,7 +6,9 @@ visual snapshots of the video's keyframes, helping users preview content
 and identify important moments without watching the entire video.

 """
-import ffmpeg, os
+
+import ffmpeg
+import os
 from loguru import logger

 from auto_archiver.core import Enricher
@@ -18,7 +20,7 @@ class ThumbnailEnricher(Enricher):
    """
    Generates thumbnails for all the media
    """
-    
+
    def enrich(self, to_enrich: Metadata) -> None:
        """
        Uses or reads the video duration to generate thumbnails
@@ -36,7 +38,9 @@ class ThumbnailEnricher(Enricher):
                if duration is None:
                    try:
                        probe = ffmpeg.probe(m.filename)
-                        duration = float(next(stream for stream in probe['streams'] if stream['codec_type'] == 'video')['duration'])
+                        duration = float(
+                            next(stream for stream in probe["streams"] if stream["codec_type"] == "video")["duration"]
+                        )
                        to_enrich.media[m_id].set("duration", duration)
                    except Exception as e:
                        logger.error(f"error getting duration of video {m.filename}: {e}")
@@ -48,11 +52,13 @@ class ThumbnailEnricher(Enricher):
                thumbnails_media = []
                for index, timestamp in enumerate(timestamps):
                    output_path = os.path.join(folder, f"out{index}.jpg")
-                    ffmpeg.input(m.filename, ss=timestamp).filter('scale', 512, -1).output(output_path, vframes=1, loglevel="quiet").run()
+                    ffmpeg.input(m.filename, ss=timestamp).filter("scale", 512, -1).output(
+                        output_path, vframes=1, loglevel="quiet"
+                    ).run()

                    try:
-                        thumbnails_media.append(Media(
-                            filename=output_path)
+                        thumbnails_media.append(
+                            Media(filename=output_path)
                            .set("id", f"thumbnail_{index}")
                            .set("timestamp", "%.3fs" % timestamp)
                        )
--- a/src/auto_archiver/modules/timestamping_enricher/manifest.py
+++ b/src/auto_archiver/modules/timestamping_enricher/manifest.py
@@ -3,38 +3,29 @@
    "type": ["enricher"],
    "requires_setup": True,
    "dependencies": {
-        "python": [
-            "loguru",
-            "slugify",
-            "tsp_client",
-            "asn1crypto",
-            "certvalidator",
-            "certifi"
-        ],
+        "python": ["loguru", "slugify", "tsp_client", "asn1crypto", "certvalidator", "certifi"],
    },
    "configs": {
        "tsa_urls": {
            "default": [
-                    # [Adobe Approved Trust List] and [Windows Cert Store]
-                    "http://timestamp.digicert.com",
-                    "http://timestamp.identrust.com",
-                    # "https://timestamp.entrust.net/TSS/RFC3161sha2TS", # not valid for timestamping
-                    # "https://timestamp.sectigo.com", # wait 15 seconds between each request.
-
-                    # [Adobe: European Union Trusted Lists].
-                    # "https://timestamp.sectigo.com/qualified", # wait 15 seconds between each request.
-
-                    # [Windows Cert Store]
-                    "http://timestamp.globalsign.com/tsa/r6advanced1",
-                    # [Adobe: European Union Trusted Lists] and [Windows Cert Store]
-                    # "http://ts.quovadisglobal.com/eu", # not valid for timestamping
-                    # "http://tsa.belgium.be/connect", # self-signed certificate in certificate chain
-                    # "https://timestamp.aped.gov.gr/qtss", # self-signed certificate in certificate chain
-                    # "http://tsa.sep.bg", # self-signed certificate in certificate chain
-                    # "http://tsa.izenpe.com", #unable to get local issuer certificate
-                    # "http://kstamp.keynectis.com/KSign", # unable to get local issuer certificate
-                    "http://tss.accv.es:8318/tsa",
-                ],
+                # [Adobe Approved Trust List] and [Windows Cert Store]
+                "http://timestamp.digicert.com",
+                "http://timestamp.identrust.com",
+                # "https://timestamp.entrust.net/TSS/RFC3161sha2TS", # not valid for timestamping
+                # "https://timestamp.sectigo.com", # wait 15 seconds between each request.
+                # [Adobe: European Union Trusted Lists].
+                # "https://timestamp.sectigo.com/qualified", # wait 15 seconds between each request.
+                # [Windows Cert Store]
+                "http://timestamp.globalsign.com/tsa/r6advanced1",
+                # [Adobe: European Union Trusted Lists] and [Windows Cert Store]
+                # "http://ts.quovadisglobal.com/eu", # not valid for timestamping
+                # "http://tsa.belgium.be/connect", # self-signed certificate in certificate chain
+                # "https://timestamp.aped.gov.gr/qtss", # self-signed certificate in certificate chain
+                # "http://tsa.sep.bg", # self-signed certificate in certificate chain
+                # "http://tsa.izenpe.com", #unable to get local issuer certificate
+                # "http://kstamp.keynectis.com/KSign", # unable to get local issuer certificate
+                "http://tss.accv.es:8318/tsa",
+            ],
            "help": "List of RFC3161 Time Stamp Authorities to use, separate with commas if passed via the command line.",
        }
    },
@@ -50,5 +41,5 @@
    ### Notes
    - Should be run after the `hash_enricher` to ensure file hashes are available.
    - Requires internet access to interact with the configured TSAs.
-    """
+    """,
 }
--- a/src/auto_archiver/modules/timestamping_enricher/timestamping_enricher.py
+++ b/src/auto_archiver/modules/timestamping_enricher/timestamping_enricher.py
@@ -11,6 +11,7 @@ import certifi
 from auto_archiver.core import Enricher
 from auto_archiver.core import Metadata, Media

+
 class TimestampingEnricher(Enricher):
    """
    Uses several RFC3161 Time Stamp Authorities to generate a timestamp token that will be preserved. This can be used to prove that a certain file existed at a certain time, useful for legal purposes, for example, to prove that a certain file was not tampered with after a certain date.
@@ -25,27 +26,30 @@ class TimestampingEnricher(Enricher):
        logger.debug(f"RFC3161 timestamping existing files for {url=}")

        # create a new text file with the existing media hashes
-        hashes = [m.get("hash").replace("SHA-256:", "").replace("SHA3-512:", "") for m in to_enrich.media if m.get("hash")]
+        hashes = [
+            m.get("hash").replace("SHA-256:", "").replace("SHA3-512:", "") for m in to_enrich.media if m.get("hash")
+        ]

        if not len(hashes):
            logger.warning(f"No hashes found in {url=}")
            return
-        
+
        tmp_dir = self.tmp_dir
        hashes_fn = os.path.join(tmp_dir, "hashes.txt")

        data_to_sign = "\n".join(hashes)
-        with open(hashes_fn, "w") as f: 
+        with open(hashes_fn, "w") as f:
            f.write(data_to_sign)
        hashes_media = Media(filename=hashes_fn)

        timestamp_tokens = []
        from slugify import slugify
+
        for tsa_url in self.tsa_urls:
            try:
                signing_settings = SigningSettings(tsp_server=tsa_url, digest_algorithm=DigestAlgorithm.SHA256)
                signer = TSPSigner()
-                message = bytes(data_to_sign, encoding='utf8')
+                message = bytes(data_to_sign, encoding="utf8")
                # send TSQ and get TSR from the TSA server
                signed = signer.sign(message=message, signing_settings=signing_settings)
                # fail if there's any issue with the certificates, uses certifi list of trusted CAs
@@ -54,7 +58,8 @@ class TimestampingEnricher(Enricher):
                cert_chain = self.download_and_verify_certificate(signed)
                # continue with saving the timestamp token
                tst_fn = os.path.join(tmp_dir, f"timestamp_token_{slugify(tsa_url)}")
-                with open(tst_fn, "wb") as f: f.write(signed)
+                with open(tst_fn, "wb") as f:
+                    f.write(signed)
                timestamp_tokens.append(Media(filename=tst_fn).set("tsa", tsa_url).set("cert_chain", cert_chain))
            except Exception as e:
                logger.warning(f"Error while timestamping {url=} with {tsa_url=}: {e}")
@@ -75,7 +80,7 @@ class TimestampingEnricher(Enricher):
        tst = ContentInfo.load(signed)

        trust_roots = []
-        with open(certifi.where(), 'rb') as f:
+        with open(certifi.where(), "rb") as f:
            for _, _, der_bytes in pem.unarmor(f.read(), multiple=True):
                trust_roots.append(der_bytes)
        context = ValidationContext(trust_roots=trust_roots)
@@ -83,11 +88,11 @@ class TimestampingEnricher(Enricher):
        certificates = tst["content"]["certificates"]
        first_cert = certificates[0].dump()
        intermediate_certs = []
-        for i in range(1, len(certificates)): # cannot use list comprehension [1:]
+        for i in range(1, len(certificates)):  # cannot use list comprehension [1:]
            intermediate_certs.append(certificates[i].dump())

        validator = CertificateValidator(first_cert, intermediate_certs=intermediate_certs, validation_context=context)
-        path = validator.validate_usage({'digital_signature'}, extended_key_usage={'time_stamping'})
+        path = validator.validate_usage({"digital_signature"}, extended_key_usage={"time_stamping"})

        cert_chain = []
        for cert in path:
@@ -96,4 +101,4 @@ class TimestampingEnricher(Enricher):
                f.write(cert.dump())
            cert_chain.append(Media(filename=cert_fn).set("subject", cert.subject.native["common_name"]))

-        return cert_chain
+        return cert_chain
--- a/src/auto_archiver/modules/twitter_api_extractor/init.py
+++ b/src/auto_archiver/modules/twitter_api_extractor/init.py
@@ -1 +1 @@
-from .twitter_api_extractor import TwitterApiExtractor
+from .twitter_api_extractor import TwitterApiExtractor
--- a/src/auto_archiver/modules/twitter_api_extractor/manifest.py
+++ b/src/auto_archiver/modules/twitter_api_extractor/manifest.py
@@ -3,21 +3,28 @@
    "type": ["extractor"],
    "requires_setup": True,
    "dependencies": {
-        "python": ["requests",
-                   "loguru",
-                   "pytwitter",
-                   "slugify",],
-        "bin": [""]
+        "python": [
+            "requests",
+            "loguru",
+            "pytwitter",
+            "slugify",
+        ],
+        "bin": [""],
    },
    "configs": {
-            "bearer_token": {"default": None, "help": "[deprecated: see bearer_tokens] twitter API bearer_token which is enough for archiving, if not provided you will need consumer_key, consumer_secret, access_token, access_secret"},
-            "bearer_tokens": {"default": [], "help": " a list of twitter API bearer_token which is enough for archiving, if not provided you will need consumer_key, consumer_secret, access_token, access_secret, if provided you can still add those for better rate limits. CSV of bearer tokens if provided via the command line",
-                              },
-            "consumer_key": {"default": None, "help": "twitter API consumer_key"},
-            "consumer_secret": {"default": None, "help": "twitter API consumer_secret"},
-            "access_token": {"default": None, "help": "twitter API access_token"},
-            "access_secret": {"default": None, "help": "twitter API access_secret"},
+        "bearer_token": {
+            "default": None,
+            "help": "[deprecated: see bearer_tokens] twitter API bearer_token which is enough for archiving, if not provided you will need consumer_key, consumer_secret, access_token, access_secret",
        },
+        "bearer_tokens": {
+            "default": [],
+            "help": " a list of twitter API bearer_token which is enough for archiving, if not provided you will need consumer_key, consumer_secret, access_token, access_secret, if provided you can still add those for better rate limits. CSV of bearer tokens if provided via the command line",
+        },
+        "consumer_key": {"default": None, "help": "twitter API consumer_key"},
+        "consumer_secret": {"default": None, "help": "twitter API consumer_secret"},
+        "access_token": {"default": None, "help": "twitter API access_token"},
+        "access_secret": {"default": None, "help": "twitter API access_secret"},
+    },
    "description": """
        The `TwitterApiExtractor` fetches tweets and associated media using the Twitter API. 
        It supports multiple API configurations for extended rate limits and reliable access. 
@@ -39,6 +46,5 @@
        - **Access Token and Secret**: Complements the consumer key for enhanced API capabilities.
        
        Credentials can be obtained by creating a Twitter developer account at [Twitter Developer Platform](https://developer.twitter.com/en).
-        """
-,
+        """,
 }
--- a/src/auto_archiver/modules/twitter_api_extractor/twitter_api_extractor.py
+++ b/src/auto_archiver/modules/twitter_api_extractor/twitter_api_extractor.py
@@ -11,8 +11,8 @@ from slugify import slugify
 from auto_archiver.core import Extractor
 from auto_archiver.core import Metadata, Media

-class TwitterApiExtractor(Extractor):

+class TwitterApiExtractor(Extractor):
    valid_url: re.Pattern = re.compile(r"(?:twitter|x).com\/(?:\#!\/)?(\w+)\/status(?:es)?\/(\d+)")

    def setup(self) -> None:
@@ -23,30 +23,38 @@ class TwitterApiExtractor(Extractor):
        if self.bearer_token:
            self.apis.append(Api(bearer_token=self.bearer_token))
        if self.consumer_key and self.consumer_secret and self.access_token and self.access_secret:
-            self.apis.append(Api(consumer_key=self.consumer_key, consumer_secret=self.consumer_secret,
-                             access_token=self.access_token, access_secret=self.access_secret))
-        assert self.api_client is not None, "Missing Twitter API configurations, please provide either AND/OR (consumer_key, consumer_secret, access_token, access_secret) to use this archiver, you can provide both for better rate-limit results."
+            self.apis.append(
+                Api(
+                    consumer_key=self.consumer_key,
+                    consumer_secret=self.consumer_secret,
+                    access_token=self.access_token,
+                    access_secret=self.access_secret,
+                )
+            )
+        assert self.api_client is not None, (
+            "Missing Twitter API configurations, please provide either AND/OR (consumer_key, consumer_secret, access_token, access_secret) to use this archiver, you can provide both for better rate-limit results."
+        )

    @property  # getter .mimetype
    def api_client(self) -> str:
        return self.apis[self.api_index]
-    
+
    def sanitize_url(self, url: str) -> str:
        # expand URL if t.co and clean tracker GET params
-        if 'https://t.co/' in url:
+        if "https://t.co/" in url:
            try:
                r = requests.get(url, timeout=30)
-                logger.debug(f'Expanded url {url} to {r.url}')
+                logger.debug(f"Expanded url {url} to {r.url}")
                url = r.url
-            except:
-                logger.error(f'Failed to expand url {url}')
+            except Exception:
+                logger.error(f"Failed to expand url {url}")
        return url

-
    def download(self, item: Metadata) -> Metadata:
        # call download retry until success or no more apis
        while self.api_index < len(self.apis):
-            if res := self.download_retry(item): return res
+            if res := self.download_retry(item):
+                return res
            self.api_index += 1
        self.api_index = 0
        return False
@@ -54,7 +62,8 @@ class TwitterApiExtractor(Extractor):
    def get_username_tweet_id(self, url):
        # detect URLs that we definitely cannot handle
        matches = self.valid_url.findall(url)
-        if not len(matches): return False, False
+        if not len(matches):
+            return False, False

        username, tweet_id = matches[0]  # only one URL supported
        logger.debug(f"Found {username=} and {tweet_id=} in {url=}")
@@ -65,10 +74,16 @@ class TwitterApiExtractor(Extractor):
        url = item.get_url()
        # detect URLs that we definitely cannot handle
        username, tweet_id = self.get_username_tweet_id(url)
-        if not username: return False
+        if not username:
+            return False

        try:
-            tweet = self.api_client.get_tweet(tweet_id, expansions=["attachments.media_keys"], media_fields=["type", "duration_ms", "url", "variants"], tweet_fields=["attachments", "author_id", "created_at", "entities", "id", "text", "possibly_sensitive"])
+            tweet = self.api_client.get_tweet(
+                tweet_id,
+                expansions=["attachments.media_keys"],
+                media_fields=["type", "duration_ms", "url", "variants"],
+                tweet_fields=["attachments", "author_id", "created_at", "entities", "id", "text", "possibly_sensitive"],
+            )
            logger.debug(tweet)
        except Exception as e:
            logger.error(f"Could not get tweet: {e}")
@@ -88,29 +103,35 @@ class TwitterApiExtractor(Extractor):
                    mimetype = "image/jpeg"
                elif hasattr(m, "variants"):
                    variant = self.choose_variant(m.variants)
-                    if not variant: continue
+                    if not variant:
+                        continue
                    media.set("src", variant.url)
                    mimetype = variant.content_type
                else:
                    continue
                logger.info(f"Found media {media}")
                ext = mimetypes.guess_extension(mimetype)
-                media.filename = self.download_from_url(media.get("src"), f'{slugify(url)}_{i}{ext}')
+                media.filename = self.download_from_url(media.get("src"), f"{slugify(url)}_{i}{ext}")
                result.add_media(media)

-        result.set_content(json.dumps({
-            "id": tweet.data.id,
-            "text": tweet.data.text,
-            "created_at": tweet.data.created_at,
-            "author_id": tweet.data.author_id,
-            "geo": tweet.data.geo,
-            "lang": tweet.data.lang,
-            "media": urls
-        }, ensure_ascii=False, indent=4))
+        result.set_content(
+            json.dumps(
+                {
+                    "id": tweet.data.id,
+                    "text": tweet.data.text,
+                    "created_at": tweet.data.created_at,
+                    "author_id": tweet.data.author_id,
+                    "geo": tweet.data.geo,
+                    "lang": tweet.data.lang,
+                    "media": urls,
+                },
+                ensure_ascii=False,
+                indent=4,
+            )
+        )
        return result.success("twitter-api")

    def choose_variant(self, variants):
-
        """
        Chooses the highest quality variable possible out of a list of variants
        """
--- a/src/auto_archiver/modules/vk_extractor/manifest.py
+++ b/src/auto_archiver/modules/vk_extractor/manifest.py
@@ -7,10 +7,8 @@
        "python": ["loguru", "vk_url_scraper"],
    },
    "configs": {
-        "username": {"required": True,
-                     "help": "valid VKontakte username"},
-        "password": {"required": True,
-                     "help": "valid VKontakte password"},
+        "username": {"required": True, "help": "valid VKontakte username"},
+        "password": {"required": True, "help": "valid VKontakte password"},
        "session_file": {
            "default": "secrets/vk_config.v2.json",
            "help": "valid VKontakte password",
--- a/src/auto_archiver/modules/vk_extractor/vk_extractor.py
+++ b/src/auto_archiver/modules/vk_extractor/vk_extractor.py
@@ -7,7 +7,7 @@ from auto_archiver.core import Metadata, Media


 class VkExtractor(Extractor):
-    """"
+    """ "
    VK videos are handled by YTDownloader, this archiver gets posts text and images.
    Currently only works for /wall posts
    """
@@ -18,11 +18,13 @@ class VkExtractor(Extractor):
    def download(self, item: Metadata) -> Metadata:
        url = item.get_url()

-        if "vk.com" not in item.netloc: return False
+        if "vk.com" not in item.netloc:
+            return False

        # some urls can contain multiple wall/photo/... parts and all will be fetched
        vk_scrapes = self.vks.scrape(url)
-        if not len(vk_scrapes): return False
+        if not len(vk_scrapes):
+            return False
        logger.debug(f"VK: got {len(vk_scrapes)} scraped instances")

        result = Metadata()
--- a/src/auto_archiver/modules/wacz_enricher/init.py
+++ b/src/auto_archiver/modules/wacz_enricher/init.py
@@ -1 +0,0 @@
-from .wacz_enricher import WaczExtractorEnricher
--- a/src/auto_archiver/modules/wacz_enricher/manifest.py
+++ b/src/auto_archiver/modules/wacz_enricher/manifest.py
@@ -1,41 +0,0 @@
-{
-    "name": "WACZ Enricher",
-    "type": ["enricher", "extractor"],
-    "entry_point": "wacz_enricher::WaczExtractorEnricher",
-    "requires_setup": True,
-    "dependencies": {
-        "python": [
-            "loguru",
-            "jsonlines",
-            "warcio"
-        ],
-        # TODO?
-        "bin": [
-            "docker"
-        ]
-    },
-    "configs": {
-            "profile": {"default": None, "help": "browsertrix-profile (for profile generation see https://github.com/webrecorder/browsertrix-crawler#creating-and-using-browser-profiles)."},
-            "docker_commands": {"default": None, "help":"if a custom docker invocation is needed"},
-            "timeout": {"default": 120, "help": "timeout for WACZ generation in seconds"},
-            "extract_media": {"default": False, "help": "If enabled all the images/videos/audio present in the WACZ archive will be extracted into separate Media and appear in the html report. The .wacz file will be kept untouched."},
-            "extract_screenshot": {"default": True, "help": "If enabled the screenshot captured by browsertrix will be extracted into separate Media and appear in the html report. The .wacz file will be kept untouched."},
-            "socks_proxy_host": {"default": None, "help": "SOCKS proxy host for browsertrix-crawler, use in combination with socks_proxy_port. eg: user:password@host"},
-            "socks_proxy_port": {"default": None, "help": "SOCKS proxy port for browsertrix-crawler, use in combination with socks_proxy_host. eg 1234"},
-            "proxy_server": {"default": None, "help": "SOCKS server proxy URL, in development"},
-        },
-    "description": """
-    Creates .WACZ archives of web pages using the `browsertrix-crawler` tool, with options for media extraction and screenshot saving.
-    [Browsertrix-crawler](https://crawler.docs.browsertrix.com/user-guide/) is a headless browser-based crawler that archives web pages in WACZ format.
-
-    ### Features
-    - Archives web pages into .WACZ format using Docker or direct invocation of `browsertrix-crawler`.
-    - Supports custom profiles for archiving private or dynamic content.
-    - Extracts media (images, videos, audio) and screenshots from the archive, optionally adding them to the enrichment pipeline.
-    - Generates metadata from the archived page's content and structure (e.g., titles, text).
-
-    ### Notes
-    - Requires Docker for running `browsertrix-crawler` .
-    - Configurable via parameters for timeout, media extraction, screenshots, and proxy settings.
-    """
-}
--- a/src/auto_archiver/modules/wacz_extractor_enricher/init.py
+++ b/src/auto_archiver/modules/wacz_extractor_enricher/init.py
@@ -0,0 +1 @@
+from .wacz_extractor_enricher import WaczExtractorEnricher
--- a/src/auto_archiver/modules/wacz_extractor_enricher/manifest.py
+++ b/src/auto_archiver/modules/wacz_extractor_enricher/manifest.py
@@ -0,0 +1,53 @@
+{
+    "name": "WACZ Enricher (and Extractor)",
+    "type": ["enricher", "extractor"],
+    "entry_point": "wacz_extractor_enricher::WaczExtractorEnricher",
+    "requires_setup": True,
+    "dependencies": {
+        "python": ["loguru", "jsonlines", "warcio"],
+        # TODO?
+        "bin": ["docker"],
+    },
+    "configs": {
+        "profile": {
+            "default": None,
+            "help": "browsertrix-profile (for profile generation see https://github.com/webrecorder/browsertrix-crawler#creating-and-using-browser-profiles).",
+        },
+        "docker_commands": {"default": None, "help": "if a custom docker invocation is needed"},
+        "timeout": {"default": 120, "help": "timeout for WACZ generation in seconds", "type": "int"},
+        "extract_media": {
+            "default": False,
+            "type": "bool",
+            "help": "If enabled all the images/videos/audio present in the WACZ archive will be extracted into separate Media and appear in the html report. The .wacz file will be kept untouched.",
+        },
+        "extract_screenshot": {
+            "default": True,
+            "type": "bool",
+            "help": "If enabled the screenshot captured by browsertrix will be extracted into separate Media and appear in the html report. The .wacz file will be kept untouched.",
+        },
+        "socks_proxy_host": {
+            "default": None,
+            "help": "SOCKS proxy host for browsertrix-crawler, use in combination with socks_proxy_port. eg: user:password@host",
+        },
+        "socks_proxy_port": {
+            "default": None,
+            "type": "int",
+            "help": "SOCKS proxy port for browsertrix-crawler, use in combination with socks_proxy_host. eg 1234",
+        },
+        "proxy_server": {"default": None, "help": "SOCKS server proxy URL, in development"},
+    },
+    "description": """
+    Creates .WACZ archives of web pages using the `browsertrix-crawler` tool, with options for media extraction and screenshot saving.
+    [Browsertrix-crawler](https://crawler.docs.browsertrix.com/user-guide/) is a headless browser-based crawler that archives web pages in WACZ format.
+
+    ### Features
+    - Archives web pages into .WACZ format using Docker or direct invocation of `browsertrix-crawler`.
+    - Supports custom profiles for archiving private or dynamic content.
+    - Extracts media (images, videos, audio) and screenshots from the archive, optionally adding them to the enrichment pipeline.
+    - Generates metadata from the archived page's content and structure (e.g., titles, text).
+
+    ### Notes
+    - Requires Docker for running `browsertrix-crawler` .
+    - Configurable via parameters for timeout, media extraction, screenshots, and proxy settings.
+    """,
+}
--- a/Show More
+++ b/Show More
				`@@ -0,0 +1 @@`
				`from .atlos_feeder_db_storage import AtlosFeederDbStorage`
				`@@ -1 +0,0 @@`
				`from .wacz_enricher import WaczExtractorEnricher`
				`@@ -0,0 +1 @@`
				`from .wacz_extractor_enricher import WaczExtractorEnricher`