Merge pull request #76 from bellingcat/dockerize

2026-06-11 12:48:28 +03:00 · 2023-05-11 13:57:37 +01:00
parent f0f844a569 cc03ad7c49
commit eb82936a04
102 changed files with 5249 additions and 2668 deletions
--- a/.dockerignore
+++ b/.dockerignore
@@ -0,0 +1,17 @@
+logs/
+browsertrix-tmp/
+tmp*/
+temp/
+.DS_Store
+__pycache__/
+local_archive/
+config*.json
+config.json
+*.env
+credentials.json
+secrets/
+instaloader/
+instaloader.session
+vk_config*.json
+anon*
+geckodriver.log
--- a/.github/workflows/docker-publish.yaml
+++ b/.github/workflows/docker-publish.yaml
@@ -0,0 +1,57 @@
+name: Docker
+
+# This workflow uses actions that are not certified by GitHub.
+# They are provided by a third-party and are governed by
+# separate terms of service, privacy policy, and support
+# documentation.
+
+on:
+  release:
+    types: [published]
+  push:
+    branches: [ "dockerize" ]
+    tags: [ "v*.*.*" ]
+
+env:
+  # Use docker.io for Docker Hub if empty
+  REGISTRY: ghcr.io
+  # github.repository as <account>/<repo>
+  IMAGE_NAME: ${{ github.repository }}
+
+
+jobs:
+  push_to_registry:
+    name: Push Docker image to Docker Hub
+    runs-on: ubuntu-latest
+    steps:
+      - name: Check out the repo
+        uses: actions/checkout@v3
+        
+      - name: Set up QEMU
+        uses: docker/setup-qemu-action@v1
+      # https://github.com/docker/setup-buildx-action
+      
+      - name: Set up Docker Buildx
+        id: buildx
+        uses: docker/setup-buildx-action@v1
+      
+      - name: Log in to Docker Hub
+        uses: docker/login-action@f054a8b539a109f9f41c372932f1ae047eff08c9
+        with:
+          username: ${{ secrets.DOCKER_USERNAME }}
+          password: ${{ secrets.DOCKER_PASSWORD }}
+      
+      - name: Extract metadata (tags, labels) for Docker
+        id: meta
+        uses: docker/metadata-action@98669ae865ea3cffbcbaa878cf57c20bbf1c6c38
+        with:
+          images: bellingcat/auto-archiver
+      
+      - name: Build and push Docker image
+        uses: docker/build-push-action@v2
+        with:
+          context: .
+          platforms: linux/amd64,linux/arm64
+          push: ${{ github.event_name != 'pull_request' }}
+          tags: ${{ steps.meta.outputs.tags }}
+          labels: ${{ steps.meta.outputs.labels }}
--- a/.github/workflows/python-publish.yaml
+++ b/.github/workflows/python-publish.yaml
@@ -0,0 +1,53 @@
+# This workflow will upload a Python Package using Twine when a release is created
+# For more information see: https://help.github.com/en/actions/language-and-framework-guides/using-python-with-github-actions#publishing-to-package-registries
+
+# This workflow uses actions that are not certified by GitHub.
+# They are provided by a third-party and are governed by
+# separate terms of service, privacy policy, and support
+# documentation.
+
+name: Pypi
+
+on:
+  release:
+    types: [published]
+  push:
+    branches: [ "dockerize" ]
+    tags: [ "v*.*.*" ]
+
+permissions:
+  contents: read
+
+jobs:
+  deploy:
+    name: Publish python package
+    runs-on: ubuntu-latest
+
+    steps:
+    - uses: actions/checkout@v3
+
+    - name: Set up Python 3.10
+      uses: actions/setup-python@v4
+      with:
+        python-version: "3.10"
+
+    - name: Install dependencies
+      run: |
+        python -m pip install --upgrade --upgrade-strategy=eager pip setuptools wheel twine pipenv
+        python -m pip install -e . --upgrade
+        python -m pipenv install --dev --python 3.10
+      env:
+        PIPENV_DEFAULT_PYTHON_VERSION: "3.10"
+
+    - name: Build wheels
+      run: |
+        python -m pipenv run python setup.py sdist bdist_wheel
+
+    - name: Publish a Python distribution to PyPI
+      uses: pypa/gh-action-pypi-publish@release/v1
+      with:
+        user: __token__
+        verbose: true
+        skip_existing: true
+        password: ${{ secrets.PYPI_API_TOKEN }}
+        packages_dir: dist/
--- a/.gitignore
+++ b/.gitignore
@@ -19,6 +19,12 @@ local_archive/
 vk_config*.json
 gd-token.json
 credentials.json
-secrets/*
+secrets*
 browsertrix/*
-browsertrix-tmp/*
+browsertrix-tmp/*
+instaloader/*
+instaloader.session
+orchestration.yaml
+auto_archiver.egg-info*
+logs*
+*.csv
--- a/36
+++ b/36
@@ -0,0 +1,36 @@
+FROM webrecorder/browsertrix-crawler:latest
+
+ENV RUNNING_IN_DOCKER=1
+
+WORKDIR /app
+
+# TODO: use custom ffmpeg builds instead of apt-get install
+RUN pip install --upgrade pip && \
+	pip install pipenv && \
+	add-apt-repository ppa:mozillateam/ppa && \
+	apt-get update && \
+	apt-get install -y gcc ffmpeg fonts-noto && \
+	apt-get install -y --no-install-recommends firefox-esr && \
+	ln -s /usr/bin/firefox-esr /usr/bin/firefox && \
+	wget https://github.com/mozilla/geckodriver/releases/download/v0.33.0/geckodriver-v0.33.0-linux64.tar.gz && \
+	tar -xvzf geckodriver* -C /usr/local/bin && \
+	chmod +x /usr/local/bin/geckodriver && \
+	rm geckodriver-v*
+
+
+# TODO: avoid copying unnecessary files, including .git
+COPY Pipfile* ./
+RUN pipenv install
+
+# doing this at the end helps during development, builds are quick
+COPY ./src/ . 
+
+# TODO: figure out how to make volumes not be root, does it depend on host or dockerfile?
+# RUN useradd --system --groups sudo --shell /bin/bash archiver && chown -R archiver:sudo .
+# USER archiver
+
+
+ENTRYPOINT ["pipenv", "run", "python3", "-m", "auto_archiver"]
+
+# should be executed with 2 volumes (3 if local_storage is used)
+# docker run --rm -v $PWD/secrets:/app/secrets -v $PWD/local_archive:/app/local_archive aa pipenv run python3 -m auto_archiver --config secrets/orchestration.yaml
--- a/19
+++ b/19
@@ -14,7 +14,6 @@ loguru = "*"
 ffmpeg-python = "*"
 selenium = "*"
 snscrape = "*"
-yt-dlp = "*"
 telethon = "*"
 google-api-python-client = "*"
 google-auth-httplib2 = "*"
@@ -23,8 +22,22 @@ oauth2client = "*"
 python-slugify = "*"
 pyyaml = "*"
 dateparser = "*"
-vk-url-scraper = "*"
 python-twitter-v2 = "*"
+instaloader = "*"
+tqdm = "*"
+jinja2 = "*"
+cryptography = "==38.0.4"
+dataclasses-json = "*"
+yt-dlp = ">=2023.2.17"
+vk-url-scraper = "*"
+uwsgi = "*"
+requests = {extras = ["socks"], version = "*"}
+# wacz = "==0.4.8"
+pywb = ">=2.7.3"

 [requires]
-python_version = "3.9"
+python_version = "3.10"
+
+[dev-packages]
+autopep8 = "*"
+setuptools-pipfile = "*"
--- a/Pipfile.lock
+++ b/Pipfile.lock
--- a/README.md
+++ b/README.md
@@ -1,186 +1,243 @@
-# Auto Archiver
+<h1 align="center">Auto Archiver</h1>
+
+[![PyPI version](https://badge.fury.io/py/auto-archiver.svg)](https://badge.fury.io/py/auto-archiver)
+[![Docker Image Version (latest by date)](https://img.shields.io/docker/v/bellingcat/auto-archiver?label=version&logo=docker)](https://hub.docker.com/r/bellingcat/auto-archiver)
+<!-- ![Docker Pulls](https://img.shields.io/docker/pulls/bellingcat/auto-archiver) -->
+<!-- [![PyPI download month](https://img.shields.io/pypi/dm/auto-archiver.svg)](https://pypi.python.org/pypi/auto-archiver/) -->
+<!-- [![Documentation Status](https://readthedocs.org/projects/vk-url-scraper/badge/?version=latest)](https://vk-url-scraper.readthedocs.io/en/latest/?badge=latest) -->
+
+
 Read the [article about Auto Archiver on bellingcat.com](https://www.bellingcat.com/resources/2022/09/22/preserve-vital-online-content-with-bellingcats-auto-archiver-tool/).


-Python script to automatically archive social media posts, videos, and images from a Google Sheets document. Uses different archivers depending on the platform, and can save content to local storage, S3 bucket (Digital Ocean Spaces, AWS, ...), and Google Drive. The Google Sheets where the links come from is updated with information about the archived content. It can be run manually or on an automated basis.
+Python tool to automatically archive social media posts, videos, and images from a Google Sheets, the console, and more. Uses different archivers depending on the platform, and can save content to local storage, S3 bucket (Digital Ocean Spaces, AWS, ...), and Google Drive. If using Google Sheets as the source for links, it will be updated with information about the archived content. It can be run manually or on an automated basis.

-## Setup
+There are 3 ways to use the auto-archiver:
+1. (easiest installation) via docker
+2. (local python install) `pip install auto-archiver`
+3. (legacy/development) clone and manually install from repo (see legacy [tutorial video](https://youtu.be/VfAhcuV2tLQ))

-Check this [tutorial video](https://youtu.be/VfAhcuV2tLQ).
+But **you always need a configuration/orchestration file**, which is where you'll configure where/what/how to archive. Make sure you read [orchestration](#orchestration).


+## How to install and run the auto-archiver

-If you are using `pipenv` (recommended), `pipenv install` is sufficient to install Python prerequisites.
+### Option 1 - docker

-You also need:
-1. [A Google Service account is necessary for use with `gspread`.](https://gspread.readthedocs.io/en/latest/oauth2.html#for-bots-using-service-account) Credentials for this account should be stored in `service_account.json`, in the same directory as the script.
-2. [ffmpeg](https://www.ffmpeg.org/) must also be installed locally for this tool to work. 
-3. [firefox](https://www.mozilla.org/en-US/firefox/new/) and [geckodriver](https://github.com/mozilla/geckodriver/releases) on a path folder like `/usr/local/bin`. 
-4. [fonts-noto](https://fonts.google.com/noto) to deal with multiple unicode characters during selenium/geckodriver's screenshots: `sudo apt install fonts-noto -y`. 
-5. Internet Archive credentials can be retrieved from https://archive.org/account/s3.php.
-6. If you would like to take archival [WACZ](https://specs.webrecorder.net/wacz/1.1.1/) snapshots using [browsertrix-crawler](https://github.com/webrecorder/browsertrix-crawler)
-   in addition to screenshots you will need to install [Docker](https://www.docker.com/).
+[![dockeri.co](https://dockerico.blankenship.io/image/bellingcat/auto-archiver)](https://hub.docker.com/r/bellingcat/auto-archiver)

-### Configuration file
-Configuration is done via a config.yaml file (see [example.config.yaml](example.config.yaml)) and some properties of that file can be overwritten via command line arguments. Here is the current result from running the `python auto_archive.py --help`:
-
-<details><summary><code>python auto_archive.py --help</code></summary>
+Docker works like a virtual machine running inside your computer, it isolates everything and makes installation simple. Since it is an isolated environment when you need to pass it your orchestration file or get downloaded media out of docker you will need to connect folders on your machine with folders inside docker with the `-v` volume flag.


+1. install [docker](https://docs.docker.com/get-docker/)
+2. pull the auto-archiver docker [image](https://hub.docker.com/r/bellingcat/auto-archiver) with `docker pull bellingcat/auto-archiver`
+3. run the docker image locally in a container: `docker run --rm -v $PWD/secrets:/app/secrets -v $PWD/local_archive:/app/local_archive bellingcat/auto-archiver --config secrets/orchestration.yaml` breaking this command down:
+   1. `docker run` tells docker to start a new container (an instance of the image)
+   2. `--rm` makes sure this container is removed after execution (less garbage locally)
+   3. `-v $PWD/secrets:/app/secrets` - your secrets folder
+      1. `-v` is a volume flag which means a folder that you have on your computer will be connected to a folder inside the docker container
+      2. `$PWD/secrets` points to a `secrets/` folder in your current working directory (where your console points to), we use this folder as a best practice to hold all the secrets/tokens/passwords/... you use
+      3. `/app/secrets` points to the path the docker container where this image can be found
+   4.  `-v $PWD/local_archive:/app/local_archive` - (optional) if you use local_storage
+       1.  `-v` same as above, this is a volume instruction
+       2.  `$PWD/local_archive` is a folder `local_archive/` in case you want to archive locally and have the files accessible outside docker
+       3.  `/app/local_archive` is a folder inside docker that you can reference in your orchestration.yml file 

-```js
-usage: auto_archive.py [-h] [--config CONFIG] [--storage {s3,local,gd}] [--sheet SHEET] [--header HEADER] [--check-if-exists] [--save-logs] [--s3-private] [--col-url URL] [--col-status STATUS] [--col-folder FOLDER]
-                       [--col-archive ARCHIVE] [--col-date DATE] [--col-thumbnail THUMBNAIL] [--col-thumbnail_index THUMBNAIL_INDEX] [--col-timestamp TIMESTAMP] [--col-title TITLE] [--col-duration DURATION]
-                       [--col-screenshot SCREENSHOT] [--col-hash HASH]
+### Option 2 - python package

-Automatically archive social media posts, videos, and images from a Google Sheets document. 
-The command line arguments will always override the configurations in the provided YAML config file (--config), only some high-level options
-are allowed via the command line and the YAML configuration file is the preferred method. The sheet must have the "url" and "status" for the archiver to work.
+<details><summary><code>Python package instructions</code></summary>
+
+1. make sure you have python 3.8 or higher installed
+2. install the package `pip/pipenv/conda install auto-archiver`
+3. test it's installed with `auto-archiver --help`
+4. run it with your orchestration file and pass any flags you want in the command line `auto-archiver --config secrets/orchestration.yaml` if your orchestration file is inside a `secrets/`, which we advise
+   
+You will also need [ffmpeg](https://www.ffmpeg.org/), [firefox](https://www.mozilla.org/en-US/firefox/new/) and [geckodriver](https://github.com/mozilla/geckodriver/releases), and optionally [fonts-noto](https://fonts.google.com/noto). Similar to the local installation. 
+
+</details>
+
+
+### Option 3 - local installation
+This can also be used for development.
+
+<details><summary><code>Legacy instructions, only use if docker/package is not an option</code></summary>
+
+
+Install the following locally:
+1. [ffmpeg](https://www.ffmpeg.org/) must also be installed locally for this tool to work. 
+2. [firefox](https://www.mozilla.org/en-US/firefox/new/) and [geckodriver](https://github.com/mozilla/geckodriver/releases) on a path folder like `/usr/local/bin`. 
+3. (optional) [fonts-noto](https://fonts.google.com/noto) to deal with multiple unicode characters during selenium/geckodriver's screenshots: `sudo apt install fonts-noto -y`. 
+
+Clone and run:
+1. `git clone https://github.com/bellingcat/auto-archiver`
+2. `pipenv install`
+3. `pipenv run python -m src.auto_archiver --config secrets/orchestration.yaml`

-optional arguments:
-  -h, --help            show this help message and exit
-  --config CONFIG       the filename of the YAML configuration file (defaults to 'config.yaml')
-  --storage {s3,local,gd}
-                        which storage to use [execution.storage in config.yaml]
-  --sheet SHEET         the name of the google sheets document [execution.sheet in config.yaml]
-  --header HEADER       1-based index for the header row [execution.header in config.yaml]
-  --check-if-exists     when possible checks if the URL has been archived before and does not archive the same URL twice [exceution.check_if_exists]
-  --save-logs           creates or appends execution logs to files logs/LEVEL.log [exceution.save_logs]
-  --s3-private          Store content without public access permission (only for storage=s3) [secrets.s3.private in config.yaml]
-  --col-url URL         the name of the column to READ url FROM (default='link')
-  --col-status STATUS   the name of the column to FILL WITH status (default='archive status')
-  --col-folder FOLDER   the name of the column to READ folder FROM (default='destination folder')
-  --col-archive ARCHIVE
-                        the name of the column to FILL WITH archive (default='archive location')
-  --col-date DATE       the name of the column to FILL WITH date (default='archive date')
-  --col-thumbnail THUMBNAIL
-                        the name of the column to FILL WITH thumbnail (default='thumbnail')
-  --col-thumbnail_index THUMBNAIL_INDEX
-                        the name of the column to FILL WITH thumbnail_index (default='thumbnail index')
-  --col-timestamp TIMESTAMP
-                        the name of the column to FILL WITH timestamp (default='upload timestamp')
-  --col-title TITLE     the name of the column to FILL WITH title (default='upload title')
-  --col-duration DURATION
-                        the name of the column to FILL WITH duration (default='duration')
-  --col-screenshot SCREENSHOT
-                        the name of the column to FILL WITH screenshot (default='screenshot')
-  --col-hash HASH       the name of the column to FILL WITH hash (default='hash')
-```

 </details><br/>

-#### Example invocations
-All the configurations can be specified in the YAML config file, but sometimes it is useful to override only some of those like the sheet that we are running the archival on, here are some examples (possibly prepended by `pipenv run`):
+# Orchestration
+The archiver work is orchestrated by the following workflow (we call each a **step**): 
+1. **Feeder** gets the links (from a spreadsheet, from the console, ...)
+2. **Archiver** tries to archive the link (twitter, youtube, ...)
+3. **Enricher** adds more info to the content (hashes, thumbnails, ...)
+4. **Formatter** creates a report from all the archived content (HTML, PDF, ...)
+5. **Database** knows what's been archived and also stores the archive result (spreadsheet, CSV, or just the console)
+
+To setup an auto-archiver instance create an `orchestration.yaml` which contains the workflow you would like. We advise you put this file into a `secrets/` folder and do not share it with others because it will contain passwords and other secrets. 
+
+The structure of orchestration file is split into 2 parts: `steps` (what **steps** to use) and `configurations` (how those steps should behave), here's a simplification:
+```yaml
+# orchestration.yaml content
+steps:
+  feeder: gsheet_feeder
+  archivers: # order matters
+    - youtubedl_archiver
+  enrichers:
+    - thumbnail_enricher
+  formatter: html_formatter
+  storages:
+    - local_storage
+  databases:
+    - gsheet_db
+
+configurations:
+  gsheet_feeder:
+    sheet: "your google sheet name"
+    header: 2 # row with header for your sheet
+  # ... configurations for the other steps here ...
+```
+
+To see all available `steps` (which archivers, storages, databses, ...) exist check the [example.orchestration.yaml](example.orchestration.yaml).
+
+All the `configurations` in the `orchestration.yaml` file (you can name it differently but need to pass it in the `--config FILENAME` argument) can be seen in the console by using the `--help` flag. They can also be overwritten, for example if you are using the `cli_feeder` to archive from the command line and want to provide the URLs you should do:

 ```bash
-# all the configurations come from config.yaml
-python auto_archive.py
+auto-archiver --config secrets/orchestration.yaml --cli_feeder.urls="url1,url2,url3"
+```

-# all the configurations come from config.yaml,
-# checks if URL is not archived twice and saves logs to logs/ folder
-python auto_archive.py --check-if-exists --save_logs
+Here's the complete workflow that the auto-archiver goes through:
+```mermaid
+graph TD
+    s((start)) --> F(fa:fa-table Feeder)
+    F -->|get and clean URL| D1{fa:fa-database Database}
+    D1 -->|is already archived| e((end))
+    D1 -->|not yet archived| a(fa:fa-download Archivers)
+    a -->|got media| E(fa:fa-chart-line Enrichers)
+    E --> S[fa:fa-box-archive Storages]
+    E --> Fo(fa:fa-code Formatter)
+    Fo --> S
+    Fo -->|update database| D2(fa:fa-database Database)
+    D2 --> e
+```

-# all the configurations come from my_config.yaml
-python auto_archive.py --config my_config.yaml
+## Orchestration checklist
+Use this to make sure you help making sure you did all the required steps:
+* [ ] you have a `/secrets` folder with all your configuration files including
+  * [ ] a orchestration file eg: `orchestration.yaml` pointing to the correct location of other files
+  * [ ] (optional if you use GoogleSheets) you have a `service_account.json` (see [how-to](https://gspread.readthedocs.io/en/latest/oauth2.html#for-bots-using-service-account))
+  * [ ] (optional for telegram) a `anon.session` which appears after the 1st run where you login to telegram
+    * if you use private channels you need to add `channel_invites` and set `join_channels=true` at least once
+  * [ ] (optional for VK) a `vk_config.v2.json`
+  * [ ] (optional for using GoogleDrive storage) `gd-token.json` (see [help script](scripts/create_update_gdrive_oauth_token.py))
+  * [ ] (optional for instagram) `instaloader.session` file which appears after the 1st run and login in instagram
+  * [ ] (optional for browsertrix) `profile.tar.gz` file

-# reads the configurations but saves archived content to google drive instead
-python auto_archive.py --config my_config.yaml --storage gd
+#### Example invocations
+The recommended way to run the auto-archiver is through Docker. The invocations below will run the auto-archiver Docker image using a configuration file that you have specified

-# uses the configurations but for another google docs sheet 
+```bash
+# all the configurations come from ./secrets/orchestration.yaml
+docker run --rm -v $PWD/secrets:/app/secrets -v $PWD/local_archive:/app/local_archive bellingcat/auto-archiver --config secrets/orchestration.yaml
+# uses the same configurations but for another google docs sheet 
 # with a header on row 2 and with some different column names
-python auto_archive.py --config my_config.yaml --sheet="use it on another sheets doc" --header=2 --col-link="put urls here"
+# notice that columns is a dictionary so you need to pass it as JSON and it will override only the values provided
+docker run --rm -v $PWD/secrets:/app/secrets -v $PWD/local_archive:/app/local_archive bellingcat/auto-archiver --config secrets/orchestration.yaml --gsheet_feeder.sheet="use it on another sheets doc" --gsheet_feeder.header=2 --gsheet_feeder.columns='{"url": "link"}'
+# all the configurations come from orchestration.yaml and specifies that s3 files should be private
+docker run --rm -v $PWD/secrets:/app/secrets -v $PWD/local_archive:/app/local_archive bellingcat/auto-archiver --config secrets/orchestration.yaml --s3_storage.private=1
+```

-# all the configurations come from config.yaml and specifies that s3 files should be private
-python auto_archive.py --s3-private
+The auto-archiver can also be run locally, if pre-requisites are correctly configured. Equivalent invocations are below.
+
+```bash
+# all the configurations come from ./secrets/orchestration.yaml
+auto-archiver --config secrets/orchestration.yaml
+# uses the same configurations but for another google docs sheet 
+# with a header on row 2 and with some different column names
+# notice that columns is a dictionary so you need to pass it as JSON and it will override only the values provided
+auto-archiver --config secrets/orchestration.yaml --gsheet_feeder.sheet="use it on another sheets doc" --gsheet_feeder.header=2 --gsheet_feeder.columns='{"url": "link"}'
+# all the configurations come from orchestration.yaml and specifies that s3 files should be private
+auto-archiver --config secrets/orchestration.yaml --s3_storage.private=1
 ```

 ### Extra notes on configuration
 #### Google Drive
 To use Google Drive storage you need the id of the shared folder in the `config.yaml` file which must be shared with the service account eg `autoarchiverservice@auto-archiver-111111.iam.gserviceaccount.com` and then you can use `--storage=gd`

-#### Telethon (Telegrams API Library)
+#### Telethon + Instagram with telegram bot
 The first time you run, you will be prompted to do a authentication with the phone number associated, alternatively you can put your `anon.session` in the root.


-## Running
-The `--sheet name` property (or `execution.sheet` in the YAML file) is the name of the Google Sheet to check for URLs. 
+## Running on Google Sheets Feeder (gsheet_feeder)
+The `--gseets_feeder.sheet` property is the name of the Google Sheet to check for URLs. 
 This sheet must have been shared with the Google Service account used by `gspread`. 
-This sheet must also have specific columns (case-insensitive) in the `header` row (see `COLUMN_NAMES` in [gworksheet.py](utils/gworksheet.py)), only the `link` and `status` columns are mandatory:
-* `Link` (required): the location of the media to be archived. This is the only column that should be supplied with data initially
-* `Archive status` (required): the status of the auto archiver script. Any row with text in this column will be skipped automatically.
-* `Destination folder`: (optional) by default files are saved to a folder called `name-of-sheets-document/name-of-sheets-tab/` using this option you can organize documents into folder from the sheet. 
-* `Archive location`: the location of the archived version. For files that were not able to be auto archived, this can be manually updated.
-* `Archive date`: the date that the auto archiver script ran for this file
-* `Upload timestamp`: the timestamp extracted from the video. (For YouTube, this unfortunately does not currently include the time)
-* `Upload title`: the "title" of the video from the original source
-* `Hash`: a hash of the first video or image found
-* `Screenshot`: a screenshot taken with from a browser view of opening the page
-* in case of videos
-  * `Duration`: duration in seconds
-  * `Thumbnail`: an image thumbnail of the video (resize row height to make this more visible)
-  * `Thumbnail index`: a link to a page that shows many thumbnails for the video, useful for quickly seeing video content
+This sheet must also have specific columns (case-insensitive) in the `header` as specified in [Gsheet.configs](src/auto_archiver/utils/gsheet.py). The default names of these columns and their purpose is:

+Inputs:

-For example, for use with this spreadsheet:
+* **Link** *(required)*: the URL of the post to archive
+* **Destination folder**: custom folder for archived file (regardless of storage)

-![A screenshot of a Google Spreadsheet with column headers defined as above, and several Youtube and Twitter URLs in the "Media URL" column](docs/demo-before.png)
+Outputs:
+* **Archive status** *(required)*: Status of archive operation
+* **Archive location**: URL of archived post
+* **Archive date**: Date archived
+* **Thumbnail**: Embeds a thumbnail for the post in the spreadsheet
+* **Timestamp**: Timestamp of original post
+* **Title**: Post title
+* **Text**: Post text
+* **Screenshot**: Link to screenshot of post
+* **Hash**: Hash of archived HTML file (which contains hashes of post media)
+* **WACZ**: Link to a WACZ web archive of post
+* **ReplayWebpage**: Link to a ReplayWebpage viewer of the WACZ archive

-```pipenv run python auto_archive.py --sheet archiver-test```
+For example, this is a spreadsheet configured with all of the columns for the auto archiver and a few URLs to archive. (Note that the column names are not case sensitive.)
+
+![A screenshot of a Google Spreadsheet with column headers defined as above, and several Youtube and Twitter URLs in the "Link" column](docs/demo-before.png)
+
+Now the auto archiver can be invoked, with this command in this example: `docker run --rm -v $PWD/secrets:/app/secrets -v $PWD/local_archive:/app/local_archive bellingcat/auto-archiver:dockerize --config secrets/orchestration-global.yaml --gsheet_feeder.sheet "Auto archive test 2023-2"`. Note that the sheet name has been overridden/specified in the command line invocation.

 When the auto archiver starts running, it updates the "Archive status" column.

-![A screenshot of a Google Spreadsheet with column headers defined as above, and several Youtube and Twitter URLs in the "Media URL" column. The auto archiver has added "archive in progress" to one of the status columns.](docs/demo-progress.png)
+![A screenshot of a Google Spreadsheet with column headers defined as above, and several Youtube and Twitter URLs in the "Link" column. The auto archiver has added "archive in progress" to one of the status columns.](docs/demo-progress.png)

 The links are downloaded and archived, and the spreadsheet is updated to the following:

 ![A screenshot of a Google Spreadsheet with videos archived and metadata added per the description of the columns above.](docs/demo-after.png)

-Note that the first row is skipped, as it is assumed to be a header row (`--header=1` and you can change it if you use more rows above). Rows with an empty URL column, or a non-empty archive column are also skipped. All sheets in the document will be checked.
+Note that the first row is skipped, as it is assumed to be a header row (`--gsheet_feeder.header=1` and you can change it if you use more rows above). Rows with an empty URL column, or a non-empty archive column are also skipped. All sheets in the document will be checked.

-## Automating
+The "archive location" link contains the path of the archived file, in local storage, S3, or in Google Drive.

-The auto-archiver can be run automatically via cron. An example crontab entry that runs the archiver every minute is as follows.
+![The archive result for a link in the demo sheet.](docs/demo-archive.png)

-```* * * * * python auto_archive.py --sheet archiver-test```
+---
+## Development
+Use `python -m src.auto_archiver --config secrets/orchestration.yaml` to run from the local development environment.

-With this configuration, the archiver should archive and store all media added to the Google Sheet every 60 seconds. Of course, additional logging information, etc. might be required.
-
-# auto_auto_archiver
-
-To make it easier to set up new auto-archiver sheets, the auto-auto-archiver will look at a particular sheet and run the auto-archiver on every sheet name in column A, starting from row 11. (It starts here to support instructional text in the first rows of the sheet, as shown below.) You can simply use your default config as for `auto_archiver.py` but use `--sheet` to specify the name of the sheet that lists the names of sheets to archive.It must be shared with the same service account.
-
-![A screenshot of a Google Spreadsheet configured to show instructional text and a list of sheet names to check with auto-archiver.](docs/auto-auto.png)
-
-# Code structure
-Code is split into functional concepts:
-1. [Archivers](archivers/) - receive a URL that they try to archive
-2. [Storages](storages/) - they deal with where the archived files go
-3. [Utilities](utils/)
-   1. [GWorksheet](utils/gworksheet.py) - facilitates some of the reading/writing tasks for a Google Worksheet
-
-### Current Archivers
-Archivers are tested in a meaningful order with Wayback Machine being the failsafe, that can easily be changed in the code. 
-
-> Note: We have 2 Twitter Archivers (`TwitterArchiver`, `TwitterApiArchiver`) because one requires Twitter API V2 credentials and has better results and the other does not rely on official APIs and misses out on some content. 
-
-```mermaid
-graph TD
-    A(Archiver) -->|parent of| B(TelethonArchiver)
-    A -->|parent of| C(TiktokArchiver)
-    A -->|parent of| D(YoutubeDLArchiver)
-    A -->|parent of| E(TelegramArchiver)
-    A -->|parent of| F(TwitterArchiver)
-    A -->|parent of| G(VkArchiver)
-    A -->|parent of| H(WaybackArchiver)
-    F -->|parent of| I(TwitterApiArchiver)
-```
-### Current Storages
-```mermaid
-graph TD
-    A(BaseStorage) -->|parent of| B(S3Storage)
-    A(BaseStorage) -->|parent of| C(LocalStorage)
-    A(BaseStorage) -->|parent of| D(GoogleDriveStorage)
-```
+#### Docker development
+working with docker locally:
+  * `docker build . -t auto-archiver` to build a local image
+  * `docker run --rm -v $PWD/secrets:/app/secrets auto-archiver pipenv run python3 -m auto_archiver --config secrets/orchestration.yaml`
+    * to use local archive, also create a volume `-v` for it by adding `-v $PWD/local_archive:/app/local_archive`


+release to docker hub
+  * `docker image tag auto-archiver bellingcat/auto-archiver:latest`
+  * `docker push bellingcat/auto-archiver`

+#### RELEASE
+* update version in [version.py](src/auto_archiver/version.py)
+* run `bash ./scripts/release.sh` and confirm
+* package is automatically updated in pypi
+* docker image is automatically pushed to dockerhup
--- a/archivers/init.py
+++ b/archivers/init.py
@@ -1,10 +0,0 @@
-# we need to explicitly expose the available imports here
-from .base_archiver import Archiver, ArchiveResult
-from .telegram_archiver import TelegramArchiver
-from .telethon_archiver import TelethonArchiver
-from .tiktok_archiver import TiktokArchiver
-from .wayback_archiver import WaybackArchiver
-from .youtubedl_archiver import YoutubeDLArchiver
-from .twitter_archiver import TwitterArchiver
-from .vk_archiver import VkArchiver
-from .twitter_api_archiver import TwitterApiArchiver
--- a/archivers/base_archiver.py
+++ b/archivers/base_archiver.py
@@ -1,348 +0,0 @@
-import os, datetime, shutil, hashlib, time, requests, re, mimetypes, subprocess
-from dataclasses import dataclass
-from abc import ABC, abstractmethod
-from urllib.parse import urlparse
-from random import randrange
-
-import ffmpeg
-from loguru import logger
-from selenium.common.exceptions import TimeoutException
-from selenium.webdriver.common.by import By
-from slugify import slugify
-
-from configs import Config
-from storages import Storage
-from utils import mkdir_if_not_exists
-
-
-@dataclass
-class ArchiveResult:
-    status: str
-    cdn_url: str = None
-    thumbnail: str = None
-    thumbnail_index: str = None
-    duration: float = None
-    title: str = None
-    timestamp: datetime.datetime = None
-    screenshot: str = None
-    wacz: str = None
-    hash: str = None
-
-class Archiver(ABC):
-    name = "default"
-    retry_regex = r"retrying at (\d+)$"
-
-    def __init__(self, storage: Storage, config: Config):
-        self.storage = storage
-        self.driver = config.webdriver
-        self.hash_algorithm = config.hash_algorithm
-        self.browsertrix = config.browsertrix_config
-
-    def __str__(self):
-        return self.__class__.__name__
-
-    def __repr__(self):
-        return self.__str__()
-
-    @abstractmethod
-    def download(self, url, check_if_exists=False): pass
-
-    def get_netloc(self, url):
-        return urlparse(url).netloc
-
-    def generate_media_page_html(self, url, urls_info: dict, object, thumbnail=None):
-        """
-        Generates an index.html page where each @urls_info is displayed
-        """
-        page = f'''<html><head><title>{url}</title><meta charset="UTF-8"></head>
-            <body>
-            <h2>Archived media from {self.name}</h2>
-            <h3><a href="{url}">{url}</a></h3><ul>'''
-
-        for url_info in urls_info:
-            mime_global = self._guess_file_type(url_info["key"])
-            preview = ""
-            if mime_global == "image":
-                preview = f'<img src="{url_info["cdn_url"]}" style="max-height:200px;max-width:400px;"></img>'
-            elif mime_global == "video":
-                preview = f'<video src="{url_info["cdn_url"]}" controls style="max-height:400px;max-width:400px;"></video>'
-            page += f'''<li><a href="{url_info['cdn_url']}">{preview}{url_info['key']}</a>: {url_info['hash']}</li>'''
-
-        page += f"</ul><h2>{self.name} object data:</h2><code>{object}</code>"
-        page += f"</body></html>"
-
-        page_key = self.get_html_key(url)
-        page_filename = os.path.join(Storage.TMP_FOLDER, page_key)
-
-        with open(page_filename, "w") as f:
-            f.write(page)
-
-        page_hash = self.get_hash(page_filename)
-
-        self.storage.upload(page_filename, page_key, extra_args={
-            'ACL': 'public-read', 'ContentType': 'text/html'})
-
-        page_cdn = self.storage.get_cdn_url(page_key)
-        return (page_cdn, page_hash, thumbnail)
-
-    def _guess_file_type(self, path: str):
-        """
-        Receives a URL or filename and returns global mimetype like 'image' or 'video'
-        see https://developer.mozilla.org/en-US/docs/Web/HTTP/Basics_of_HTTP/MIME_types/Common_types
-        """
-        mime = mimetypes.guess_type(path)[0]
-        if mime is not None:
-            return mime.split("/")[0]
-        return ""
-
-    def download_from_url(self, url, to_filename):
-        headers = {
-            'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/81.0.4044.138 Safari/537.36'
-        }
-        d = requests.get(url, headers=headers)
-        with open(to_filename, 'wb') as f:
-            f.write(d.content)
-
-    def generate_media_page(self, urls, url, object):
-        """
-        For a list of media urls, fetch them, upload them
-        and call self.generate_media_page_html with them
-        """
-
-        thumbnail = None
-        uploaded_media = []
-        for media_url in urls:
-            key = self._get_key_from_url(media_url, ".jpg")
-
-            filename = os.path.join(Storage.TMP_FOLDER, key)
-            self.download_from_url(media_url, filename)
-            self.storage.upload(filename, key)
-            hash = self.get_hash(filename)
-            cdn_url = self.storage.get_cdn_url(key)
-
-            if thumbnail is None:
-                thumbnail = cdn_url
-            uploaded_media.append({'cdn_url': cdn_url, 'key': key, 'hash': hash})
-
-        return self.generate_media_page_html(url, uploaded_media, object, thumbnail=thumbnail)
-
-    def get_key(self, filename):
-        """
-        returns a key in the format "[archiverName]_[filename]" includes extension
-        """
-        tail = os.path.split(filename)[1]  # returns filename.ext from full path
-        _id, extension = os.path.splitext(tail)  # returns [filename, .ext]
-        if 'unknown_video' in _id:
-            _id = _id.replace('unknown_video', 'jpg')
-
-        # long filenames can cause problems, so trim them if necessary
-        if len(_id) > 128:
-            _id = _id[-128:]
-
-        return f'{self.name}_{_id}{extension}'
-
-    def get_html_key(self, url):
-        return self._get_key_from_url(url, ".html")
-
-    def _get_key_from_url(self, url, with_extension: str = None, append_datetime: bool = False):
-        """
-        Receives a URL and returns a slugified version of the URL path
-        if a string is passed in @with_extension the slug is appended with it if there is no "." in the slug
-        if @append_date is true, the key adds a timestamp after the URL slug and before the extension
-        """
-        url_path = urlparse(url).path
-        path, ext = os.path.splitext(url_path)
-        slug = slugify(path)
-        if append_datetime:
-            slug += "-" + slugify(datetime.datetime.utcnow().isoformat())
-        if len(ext):
-            slug += ext
-        if with_extension is not None:
-            if "." not in slug:
-                slug += with_extension
-        return self.get_key(slug)
-
-    def get_hash(self, filename):
-        with open(filename, "rb") as f:
-            bytes = f.read()  # read entire file as bytes
-            logger.debug(f'Hash algorithm is {self.hash_algorithm}')
-
-            if self.hash_algorithm == "SHA-256": hash = hashlib.sha256(bytes)
-            elif self.hash_algorithm == "SHA3-512": hash = hashlib.sha3_512(bytes)
-            else: raise Exception(f"Unknown Hash Algorithm of {self.hash_algorithm}")
-
-        return hash.hexdigest()
-
-    def get_screenshot(self, url):
-        logger.debug(f"getting screenshot for {url=}")
-        key = self._get_key_from_url(url, ".png", append_datetime=True)
-        filename = os.path.join(Storage.TMP_FOLDER, key)
-
-        # Accept cookies popup dismiss for ytdlp video
-        if 'facebook.com' in url:
-            try:
-                logger.debug(f'Trying fb click accept cookie popup for {url}')
-                self.driver.get("http://www.facebook.com")
-                foo = self.driver.find_element(By.XPATH, "//button[@data-cookiebanner='accept_only_essential_button']")
-                foo.click()
-                logger.debug(f'fb click worked')
-                # linux server needs a sleep otherwise facebook cookie won't have worked and we'll get a popup on next page
-                time.sleep(2)
-            except:
-                logger.warning(f'Failed on fb accept cookies for url {url}')
-
-        try:
-            self.driver.get(url)
-            time.sleep(6)
-        except TimeoutException:
-            logger.info("TimeoutException loading page for screenshot")
-
-        self.driver.save_screenshot(filename)
-        self.storage.upload(filename, key, extra_args={'ACL': 'public-read', 'ContentType': 'image/png'})
-
-        return self.storage.get_cdn_url(key)
-
-    def get_wacz(self, url):
-        if not self.browsertrix.enabled:
-            logger.debug(f"Browsertrix WACZ generation is not enabled, skipping.")
-            return 
-
-        logger.debug(f"getting wacz for {url}")
-        key = self._get_key_from_url(url, ".wacz", append_datetime=True)
-        collection = re.sub('[^0-9a-zA-Z]+', '', key.replace(".wacz", ""))
-
-        browsertrix_home = os.path.join(os.getcwd(), "browsertrix-tmp")
-        cmd = [
-            "docker", "run",
-            "-v", f"{browsertrix_home}:/crawls/",
-            "-it",
-            "webrecorder/browsertrix-crawler", "crawl",
-            "--url", url,
-            "--scopeType", "page",
-            "--generateWACZ",
-            "--text",
-            "--collection", collection,
-            "--behaviors", "autoscroll,autoplay,autofetch,siteSpecific",
-            "--behaviorTimeout", str(self.browsertrix.timeout_seconds),
-            "--timeout", str(self.browsertrix.timeout_seconds)
-        ]
-
-        if not os.path.isdir(browsertrix_home):
-            os.mkdir(browsertrix_home)
-
-        if self.browsertrix.profile:
-            shutil.copyfile(self.browsertrix.profile, os.path.join(browsertrix_home, "profile.tar.gz"))
-            cmd.extend(["--profile", "/crawls/profile.tar.gz"])
-
-        try:
-            logger.info(f"Running browsertrix-crawler: {' '.join(cmd)}")
-            subprocess.run(cmd, check=True)
-        except Exception as e:
-            logger.error(f"WACZ generation failed: {e}")
-            return
-
-        filename = os.path.join(browsertrix_home, "collections", collection, f"{collection}.wacz")
-
-        self.storage.upload(filename, key, extra_args={
-                            'ACL': 'public-read', 'ContentType': 'application/zip'})
-
-        # clean up the local browsertrix files
-        try:
-            shutil.rmtree(browsertrix_home)
-        except PermissionError:
-            logger.warn(f"Unable to clean up browsertrix-crawler files in {browsertrix_home}")
-
-        return self.storage.get_cdn_url(key)
-
-    def get_thumbnails(self, filename, key, duration=None):
-        thumbnails_folder = os.path.splitext(filename)[0] + os.path.sep
-        key_folder = key.split('.')[0] + os.path.sep
-
-        mkdir_if_not_exists(thumbnails_folder)
-
-        fps = 0.5
-        if duration is not None:
-            duration = float(duration)
-
-            if duration < 60:
-                fps = 10.0 / duration
-            elif duration < 120:
-                fps = 20.0 / duration
-            else:
-                fps = 40.0 / duration
-
-        stream = ffmpeg.input(filename)
-        stream = ffmpeg.filter(stream, 'fps', fps=fps).filter('scale', 512, -1)
-        stream.output(thumbnails_folder + 'out%d.jpg').run()
-
-        thumbnails = os.listdir(thumbnails_folder)
-        cdn_urls = []
-        for fname in thumbnails:
-            if fname[-3:] == 'jpg':
-                thumbnail_filename = thumbnails_folder + fname
-                key = os.path.join(key_folder, fname)
-
-                self.storage.upload(thumbnail_filename, key)
-                cdn_url = self.storage.get_cdn_url(key)
-                cdn_urls.append(cdn_url)
-
-        if len(cdn_urls) == 0:
-            return ('', '')
-
-        key_thumb = cdn_urls[int(len(cdn_urls) * 0.1)]
-
-        index_page = f'''<html><head><title>{filename}</title><meta charset="UTF-8"></head>
-            <body>'''
-
-        for t in cdn_urls:
-            index_page += f'<img src="{t}" />'
-
-        index_page += f"</body></html>"
-        index_fname = thumbnails_folder + 'index.html'
-
-        with open(index_fname, 'w') as f:
-            f.write(index_page)
-
-        thumb_index = key_folder + 'index.html'
-
-        self.storage.upload(index_fname, thumb_index, extra_args={
-                            'ACL': 'public-read', 'ContentType': 'text/html'})
-        shutil.rmtree(thumbnails_folder)
-
-        thumb_index_cdn_url = self.storage.get_cdn_url(thumb_index)
-
-        return (key_thumb, thumb_index_cdn_url)
-
-    def signal_retry_in(self, min_seconds=1800, max_seconds=7200, **kwargs):
-        """
-        sets state to retry in random between (min_seconds, max_seconds)
-        """
-        now = datetime.datetime.now().timestamp()
-        retry_at = int(now + randrange(min_seconds, max_seconds))
-        logger.debug(f"signaling {retry_at=}")
-        return ArchiveResult(status=f'retrying at {retry_at}', **kwargs)
-
-    def is_retry(status):
-        return re.search(Archiver.retry_regex, status) is not None
-
-    def should_retry_from_status(status):
-        """
-        checks status against message in signal_retry_in
-        returns true if enough time has elapsed, false otherwise
-        """
-        match = re.search(Archiver.retry_regex, status)
-        if match:
-            retry_at = int(match.group(1))
-            now = datetime.datetime.now().timestamp()
-            should_retry = now >= retry_at
-            logger.debug(f"{should_retry=} since {now=} and {retry_at=}")
-            return should_retry
-        return False
-
-    def remove_retry(status):
-        """
-        transforms the status from retry into something else
-        """
-        new_status = re.sub(Archiver.retry_regex, "failed: too many retries", status, 0)
-        logger.debug(f"removing retry message at {status=}, got {new_status=}")
-        return new_status
--- a/archivers/telegram_archiver.py
+++ b/archivers/telegram_archiver.py
@@ -1,89 +0,0 @@
-import os, requests, re
-
-import html
-from bs4 import BeautifulSoup
-from loguru import logger
-
-from .base_archiver import Archiver, ArchiveResult
-from storages import Storage
-
-
-class TelegramArchiver(Archiver):
-    name = "telegram"
-
-    def download(self, url, check_if_exists=False):
-        # detect URLs that we definitely cannot handle
-        if 't.me' != self.get_netloc(url):
-            return False
-
-        headers = {
-            'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/81.0.4044.138 Safari/537.36'
-        }
-        status = "success"
-
-        original_url = url
-
-        # TODO: check if we can do this more resilient to variable URLs
-        if url[-8:] != "?embed=1":
-            url += "?embed=1"
-
-        screenshot = self.get_screenshot(url)
-        wacz = self.get_wacz(url)
-
-        t = requests.get(url, headers=headers)
-        s = BeautifulSoup(t.content, 'html.parser')
-        video = s.find("video")
-
-        if video is None:
-            logger.warning("could not find video")
-            image_tags = s.find_all(class_="js-message_photo")
-
-            images = []
-            for im in image_tags:
-                urls = [u.replace("'", "") for u in re.findall(r'url\((.*?)\)', im['style'])]
-                images += urls
-
-            page_cdn, page_hash, thumbnail = self.generate_media_page(images, url, html.escape(str(t.content)))
-            time_elements = s.find_all('time')
-            timestamp = time_elements[0].get('datetime') if len(time_elements) else None
-
-            return ArchiveResult(status="success", cdn_url=page_cdn, screenshot=screenshot, hash=page_hash, thumbnail=thumbnail, timestamp=timestamp, wacz=wacz)
-
-        video_url = video.get('src')
-        video_id = video_url.split('/')[-1].split('?')[0]
-        key = self.get_key(video_id)
-
-        filename = os.path.join(Storage.TMP_FOLDER, key)
-
-        if check_if_exists and self.storage.exists(key):
-            status = 'already archived'
-
-        v = requests.get(video_url, headers=headers)
-
-        with open(filename, 'wb') as f:
-            f.write(v.content)
-
-        if status != 'already archived':
-            self.storage.upload(filename, key)
-
-        hash = self.get_hash(filename)
-
-        # extract duration from HTML
-        try:
-            duration = s.find_all('time')[0].contents[0]
-            if ':' in duration:
-                duration = float(duration.split(
-                    ':')[0]) * 60 + float(duration.split(':')[1])
-            else:
-                duration = float(duration)
-        except:
-            duration = ""
-
-        # process thumbnails
-        key_thumb, thumb_index = self.get_thumbnails(
-            filename, key, duration=duration)
-        os.remove(filename)
-
-        cdn_url = self.storage.get_cdn_url(key)
-        return ArchiveResult(status=status, cdn_url=cdn_url, thumbnail=key_thumb, thumbnail_index=thumb_index,
-                             duration=duration, title=original_url, timestamp=s.find_all('time')[1].get('datetime'), hash=hash, screenshot=screenshot, wacz=wacz)
--- a/archivers/telethon_archiver.py
+++ b/archivers/telethon_archiver.py
@@ -1,127 +0,0 @@
-import os, re
-
-import html
-from loguru import logger
-from telethon.sync import TelegramClient
-from telethon.errors import ChannelInvalidError
-
-from storages import Storage
-from .base_archiver import Archiver, ArchiveResult
-from configs import Config
-from utils import getattr_or
-
-
-class TelethonArchiver(Archiver):
-    name = "telethon"
-    link_pattern = re.compile(r"https:\/\/t\.me(\/c){0,1}\/(.+)\/(\d+)")
-
-    def __init__(self, storage: Storage, config: Config):
-        super().__init__(storage, config)
-        if config.telegram_config:
-            c = config.telegram_config
-            self.client = TelegramClient("./anon", c.api_id, c.api_hash)
-            self.bot_token = c.bot_token
-
-    def _get_media_posts_in_group(self, chat, original_post, max_amp=10):
-        """
-        Searches for Telegram posts that are part of the same group of uploads
-        The search is conducted around the id of the original post with an amplitude
-        of `max_amp` both ways
-        Returns a list of [post] where each post has media and is in the same grouped_id
-        """
-        if getattr_or(original_post, "grouped_id") is None:
-            return [original_post] if getattr_or(original_post, "media") else []
-
-        search_ids = [i for i in range(original_post.id - max_amp, original_post.id + max_amp + 1)]
-        posts = self.client.get_messages(chat, ids=search_ids)
-        media = []
-        for post in posts:
-            if post is not None and post.grouped_id == original_post.grouped_id and post.media is not None:
-                media.append(post)
-        return media
-
-    def download(self, url, check_if_exists=False):
-        if not hasattr(self, "client"):
-            logger.warning('Missing Telethon config')
-            return False
-
-        # detect URLs that we definitely cannot handle
-        matches = self.link_pattern.findall(url)
-        if not len(matches):
-            return False
-
-        status = "success"
-
-        # app will ask (stall for user input!) for phone number and auth code if anon.session not found
-        with self.client.start(bot_token=self.bot_token):
-            matches = list(matches[0])
-            chat, post_id = matches[1], matches[2]
-
-            post_id = int(post_id)
-
-            try:
-                post = self.client.get_messages(chat, ids=post_id)
-            except ValueError as e:
-                logger.error(f"Could not fetch telegram {url} possibly it's private: {e}")
-                return False
-            except ChannelInvalidError as e:
-                logger.error(f"Could not fetch telegram {url}. This error can be fixed if you setup a bot_token in addition to api_id and api_hash: {e}")
-                return False
-
-            if post is None: return False
-
-            media_posts = self._get_media_posts_in_group(chat, post)
-            logger.debug(f'got {len(media_posts)=} for {url=}')
-
-            screenshot = self.get_screenshot(url)
-            wacz = self.get_wacz(url)
-
-            if len(media_posts) > 0:
-                key = self.get_html_key(url)
-
-                if check_if_exists and self.storage.exists(key):
-                    # only s3 storage supports storage.exists as not implemented on gd
-                    cdn_url = self.storage.get_cdn_url(key)
-                    return ArchiveResult(status='already archived', cdn_url=cdn_url, title=post.message, timestamp=post.date, screenshot=screenshot, wacz=wacz)
-
-                key_thumb, thumb_index = None, None
-                group_id = post.grouped_id if post.grouped_id is not None else post.id
-                uploaded_media = []
-                message = post.message
-                for mp in media_posts:
-                    if len(mp.message) > len(message): message = mp.message
-
-                    # media can also be in entities
-                    if mp.entities:
-                        other_media_urls = [e.url for e in mp.entities if hasattr(e, "url") and e.url and self._guess_file_type(e.url) in ["video", "image"]]
-                        logger.debug(f"Got {len(other_media_urls)} other medial urls from {mp.id=}: {other_media_urls}")
-                        for om_url in other_media_urls:
-                            filename = os.path.join(Storage.TMP_FOLDER, f'{chat}_{group_id}_{self._get_key_from_url(om_url)}')
-                            self.download_from_url(om_url, filename)
-                            key = filename.split(Storage.TMP_FOLDER)[1]
-                            self.storage.upload(filename, key)
-                            hash = self.get_hash(filename)
-                            cdn_url = self.storage.get_cdn_url(key)
-                            uploaded_media.append({'cdn_url': cdn_url, 'key': key, 'hash': hash})
-
-                    filename_dest = os.path.join(Storage.TMP_FOLDER, f'{chat}_{group_id}', str(mp.id))
-                    filename = self.client.download_media(mp.media, filename_dest)
-                    if not filename:
-                        logger.debug(f"Empty media found, skipping {str(mp)=}")
-                        continue
-
-                    key = filename.split(Storage.TMP_FOLDER)[1]
-                    self.storage.upload(filename, key)
-                    hash = self.get_hash(filename)
-                    cdn_url = self.storage.get_cdn_url(key)
-                    uploaded_media.append({'cdn_url': cdn_url, 'key': key, 'hash': hash})
-                    if key_thumb is None:
-                        key_thumb, thumb_index = self.get_thumbnails(filename, key)
-                    os.remove(filename)
-
-                page_cdn, page_hash, _ = self.generate_media_page_html(url, uploaded_media, html.escape(str(post)))
-
-                return ArchiveResult(status=status, cdn_url=page_cdn, title=message, timestamp=post.date, hash=page_hash, screenshot=screenshot, thumbnail=key_thumb, thumbnail_index=thumb_index, wacz=wacz)
-
-            page_cdn, page_hash, _ = self.generate_media_page_html(url, [], html.escape(str(post)))
-            return ArchiveResult(status=status, cdn_url=page_cdn, title=post.message, timestamp=getattr_or(post, "date"), hash=page_hash, screenshot=screenshot, wacz=wacz)
--- a/archivers/tiktok_archiver.py
+++ b/archivers/tiktok_archiver.py
@@ -1,72 +0,0 @@
-import os, traceback
-import tiktok_downloader
-from loguru import logger
-
-from .base_archiver import Archiver, ArchiveResult
-from storages import Storage
-
-
-class TiktokArchiver(Archiver):
-    name = "tiktok"
-
-    def download(self, url, check_if_exists=False):
-        if 'tiktok.com' not in url:
-            return False
-
-        status = 'success'
-
-        try:
-            info = tiktok_downloader.info_post(url)
-            key = self.get_key(f'{info.id}.mp4')
-            filename = os.path.join(Storage.TMP_FOLDER, key)
-            logger.info(f'found video {key=}')
-
-            if check_if_exists and self.storage.exists(key):
-                status = 'already archived'
-
-            media = tiktok_downloader.snaptik(url).get_media()
-
-            if len(media) <= 0:
-                if status == 'already archived':
-                    return ArchiveResult(status='Could not download media, but already archived', cdn_url=self.storage.get_cdn_url(key))
-                else:
-                    return ArchiveResult(status='Could not download media')
-
-            logger.info(f'downloading video {key=}')
-            media[0].download(filename)
-
-            if status != 'already archived':
-                logger.info(f'uploading video {key=}')
-                self.storage.upload(filename, key)
-
-            try:
-                key_thumb, thumb_index = self.get_thumbnails(filename, key, duration=info.duration)
-            except Exception as e:
-                logger.error(e)
-                key_thumb = ''
-                thumb_index = 'error creating thumbnails'
-
-            hash = self.get_hash(filename)
-            screenshot = self.get_screenshot(url)
-            wacz = self.get_wacz(url)
-
-            try: os.remove(filename)
-            except FileNotFoundError:
-                logger.info(f'tmp file not found thus not deleted {filename}')
-            cdn_url = self.storage.get_cdn_url(key)
-            timestamp = info.create.isoformat() if hasattr(info, "create") else None
-
-            return ArchiveResult(status=status, cdn_url=cdn_url, thumbnail=key_thumb,
-                                 thumbnail_index=thumb_index, duration=getattr(info, "duration", 0), title=getattr(info, "caption", ""),
-                                 timestamp=timestamp, hash=hash, screenshot=screenshot, wacz=wacz)
-
-        except tiktok_downloader.Except.InvalidUrl as e:
-            status = 'Invalid URL'
-            logger.warning(f'Invalid URL on {url}  {e}\n{traceback.format_exc()}')
-            return ArchiveResult(status=status)
-
-        except:
-            error = traceback.format_exc()
-            status = 'Other Tiktok error: ' + str(error)
-            logger.warning(f'Other Tiktok error' + str(error))
-            return ArchiveResult(status=status)
--- a/archivers/twitter_api_archiver.py
+++ b/archivers/twitter_api_archiver.py
@@ -1,75 +0,0 @@
-
-import json
-from datetime import datetime
-from loguru import logger
-from pytwitter import Api
-
-from storages.base_storage import Storage
-from configs import Config
-from .base_archiver import ArchiveResult
-from .twitter_archiver import TwitterArchiver
-
-
-class TwitterApiArchiver(TwitterArchiver):
-    name = "twitter_api"
-
-    def __init__(self, storage: Storage, config: Config):
-        super().__init__(storage, config)
-        c = config.twitter_config
-
-        if c.bearer_token:
-            self.api = Api(bearer_token=c.bearer_token)
-        elif c.consumer_key and c.consumer_secret and c.access_token and c.access_secret:
-            self.api = Api(
-                consumer_key=c.consumer_key, consumer_secret=c.consumer_secret, access_token=c.access_token, access_secret=c.access_secret)
-
-    def download(self, url, check_if_exists=False):
-        if not hasattr(self, "api"):
-            logger.warning('Missing Twitter API config')
-            return False
-
-        username, tweet_id = self.get_username_tweet_id(url)
-        if not username: return False
-
-        tweet = self.api.get_tweet(tweet_id, expansions=["attachments.media_keys"], media_fields=["type", "duration_ms", "url", "variants"], tweet_fields=["attachments", "author_id", "created_at", "entities", "id", "text", "possibly_sensitive"])
-        timestamp = datetime.strptime(tweet.data.created_at, "%Y-%m-%dT%H:%M:%S.%fZ")
-
-        # check if exists
-        key = self.get_html_key(url)
-        if check_if_exists and self.storage.exists(key):
-            # only s3 storage supports storage.exists as not implemented on gd
-            cdn_url = self.storage.get_cdn_url(key)
-            screenshot = self.get_screenshot(url)
-            return ArchiveResult(status='already archived', cdn_url=cdn_url, title=tweet.data.text, timestamp=timestamp, screenshot=screenshot)
-
-        urls = []
-        if tweet.includes:
-            for m in tweet.includes.media:
-                if m.url:
-                    urls.append(m.url)
-                elif hasattr(m, "variants"):
-                    var_url = self.choose_variant(m.variants)
-                    urls.append(var_url)
-                else:
-                    urls.append(None)  # will trigger error
-
-            for u in urls:
-                if u is None:
-                    logger.debug(f"Should not have gotten None url for {tweet.includes.media=} so going to download_alternative in twitter_archiver")
-                    return self.download_alternative(url, tweet_id)
-        logger.debug(f"found {urls=}")
-
-        output = json.dumps({
-            "id": tweet.data.id,
-            "text": tweet.data.text,
-            "created_at": tweet.data.created_at,
-            "author_id": tweet.data.author_id,
-            "geo": tweet.data.geo,
-            "lang": tweet.data.lang,
-            "media": urls
-        }, ensure_ascii=False, indent=4)
-
-        screenshot = self.get_screenshot(url)
-        wacz = self.get_wacz(url)
-        page_cdn, page_hash, thumbnail = self.generate_media_page(urls, url, output)
-        return ArchiveResult(status="success", cdn_url=page_cdn, screenshot=screenshot, hash=page_hash, thumbnail=thumbnail, timestamp=timestamp, title=tweet.data.text, wacz=wacz)
--- a/archivers/twitter_archiver.py
+++ b/archivers/twitter_archiver.py
@@ -1,105 +0,0 @@
-import html, re, requests
-from datetime import datetime
-from loguru import logger
-from snscrape.modules.twitter import TwitterTweetScraper, Video, Gif, Photo
-
-from .base_archiver import Archiver, ArchiveResult
-
-class TwitterArchiver(Archiver):
-    """
-    This Twitter Archiver uses unofficial scraping methods, and it works as 
-    an alternative to TwitterApiArchiver when no API credentials are provided.
-    """
-
-    name = "twitter"
-    link_pattern = re.compile(r"twitter.com\/(?:\#!\/)?(\w+)\/status(?:es)?\/(\d+)")
-
-    def get_username_tweet_id(self, url):
-        # detect URLs that we definitely cannot handle
-        matches = self.link_pattern.findall(url)
-        if not len(matches): return False, False
-
-        username, tweet_id = matches[0]  # only one URL supported
-        logger.debug(f"Found {username=} and {tweet_id=} in {url=}")
-
-        return username, tweet_id
-
-    def download(self, url, check_if_exists=False):
-        username, tweet_id = self.get_username_tweet_id(url)
-        if not username: return False
-
-        scr = TwitterTweetScraper(tweet_id)
-
-        try:
-            tweet = next(scr.get_items())
-        except Exception as ex:
-            logger.warning(f"can't get tweet: {type(ex).__name__} occurred. args: {ex.args}")
-            return self.download_alternative(url, tweet_id)
-
-        if tweet.media is None:
-            logger.debug(f'No media found, archiving tweet text only')
-            screenshot = self.get_screenshot(url)
-            wacz = self.get_wacz(url)
-            page_cdn, page_hash, _ = self.generate_media_page_html(url, [], html.escape(tweet.json()))
-            return ArchiveResult(status="success", cdn_url=page_cdn, title=tweet.content, timestamp=tweet.date, hash=page_hash, screenshot=screenshot, wacz=wacz)
-
-        urls = []
-
-        for media in tweet.media:
-            if type(media) == Video:
-                variant = max(
-                    [v for v in media.variants if v.bitrate], key=lambda v: v.bitrate)
-                urls.append(variant.url)
-            elif type(media) == Gif:
-                urls.append(media.variants[0].url)
-            elif type(media) == Photo:
-                urls.append(media.fullUrl.replace('name=large', 'name=orig'))
-            else:
-                logger.warning(f"Could not get media URL of {media}")
-
-        page_cdn, page_hash, thumbnail = self.generate_media_page(urls, url, tweet.json())
-
-        screenshot = self.get_screenshot(url)
-        wacz = self.get_wacz(url)
-
-        return ArchiveResult(status="success", cdn_url=page_cdn, screenshot=screenshot, hash=page_hash, thumbnail=thumbnail, timestamp=tweet.date, title=tweet.content, wacz=wacz)
-
-    def download_alternative(self, url, tweet_id):
-        # https://stackoverflow.com/a/71867055/6196010
-        logger.debug(f"Trying twitter hack for {url=}")
-        hack_url = f"https://cdn.syndication.twimg.com/tweet?id={tweet_id}"
-        r = requests.get(hack_url)
-        if r.status_code != 200: return False
-        tweet = r.json()
-
-        urls = []
-        for p in tweet["photos"]:
-            urls.append(p["url"])
-
-        # 1 tweet has 1 video max
-        if "video" in tweet:
-            v = tweet["video"]
-            urls.append(self.choose_variant(v.get("variants", [])))
-
-        logger.debug(f"Twitter hack got {urls=}")
-
-        timestamp = datetime.strptime(tweet["created_at"], "%Y-%m-%dT%H:%M:%S.%fZ")
-        screenshot = self.get_screenshot(url)
-        wacz = self.get_wacz(url)
-        page_cdn, page_hash, thumbnail = self.generate_media_page(urls, url, r.text)
-        return ArchiveResult(status="success", cdn_url=page_cdn, screenshot=screenshot, hash=page_hash, thumbnail=thumbnail, timestamp=timestamp, title=tweet["text"], wacz=wacz)
-
-    def choose_variant(self, variants):
-        # choosing the highest quality possible
-        variant, width, height = None, 0, 0
-        for var in variants:
-            if var["type"] == "video/mp4":
-                width_height = re.search(r"\/(\d+)x(\d+)\/", var["src"])
-                if width_height:
-                    w, h = int(width_height[1]), int(width_height[2])
-                    if w > width or h > height:
-                        width, height = w, h
-                        variant = var.get("src", variant)
-            else:
-                variant = var.get("src") if not variant else variant
-        return variant
--- a/archivers/vk_archiver.py
+++ b/archivers/vk_archiver.py
@@ -1,74 +0,0 @@
-import re, json, mimetypes, os
-
-from loguru import logger
-from vk_url_scraper import VkScraper, DateTimeEncoder
-
-from storages import Storage
-from .base_archiver import Archiver, ArchiveResult
-from configs import Config
-
-
-class VkArchiver(Archiver):
-    """"
-    VK videos are handled by YTDownloader, this archiver gets posts text and images.
-    Currently only works for /wall posts
-    """
-    name = "vk"
-    wall_pattern = re.compile(r"(wall.{0,1}\d+_\d+)")
-    photo_pattern = re.compile(r"(photo.{0,1}\d+_\d+)")
-
-    def __init__(self, storage: Storage, config: Config): 
-        super().__init__(storage, config)
-        if config.vk_config != None:
-            self.vks = VkScraper(config.vk_config.username, config.vk_config.password)
-
-    def download(self, url, check_if_exists=False):
-        if not hasattr(self, "vks") or self.vks is None:
-            logger.debug("VK archiver was not supplied with credentials.")
-            return False
-
-        key = self.get_html_key(url)
-        # if check_if_exists and self.storage.exists(key):
-        #     screenshot = self.get_screenshot(url)
-        #     cdn_url = self.storage.get_cdn_url(key)
-        #     return ArchiveResult(status="already archived", cdn_url=cdn_url, screenshot=screenshot)
-
-        results = self.vks.scrape(url)  # some urls can contain multiple wall/photo/... parts and all will be fetched
-        if len(results) == 0:
-            return False
-
-        def dump_payload(p): return json.dumps(p, ensure_ascii=False, indent=4, cls=DateTimeEncoder)
-        textual_output = ""
-        title, datetime = results[0]["text"], results[0]["datetime"]
-        urls_found = []
-        for res in results:
-            textual_output += f"id: {res['id']}<br>time utc: {res['datetime']}<br>text: {res['text']}<br>payload: {dump_payload(res['payload'])}<br><hr/><br>"
-            title = res["text"] if len(title) == 0 else title
-            datetime = res["datetime"] if not datetime else datetime
-            for attachments in res["attachments"].values():
-                urls_found.extend(attachments)
-
-        # we don't call generate_media_page which downloads urls because it cannot download vk video urls
-        thumbnail, thumbnail_index = None, None
-        uploaded_media = []
-        filenames = self.vks.download_media(results, Storage.TMP_FOLDER)
-        for filename in filenames:
-            key = self.get_key(filename)
-            self.storage.upload(filename, key)
-            hash = self.get_hash(filename)
-            cdn_url = self.storage.get_cdn_url(key)
-            try:
-                _type = mimetypes.guess_type(filename)[0].split("/")[0]
-                if _type == "image" and thumbnail is None:
-                    thumbnail = cdn_url
-                if _type == "video" and (thumbnail is None or thumbnail_index is None):
-                    thumbnail, thumbnail_index = self.get_thumbnails(filename, key)
-            except Exception as e:
-                logger.warning(f"failed to get thumb for {filename=} with {e=}")
-            uploaded_media.append({'cdn_url': cdn_url, 'key': key, 'hash': hash})
-
-        page_cdn, page_hash, thumbnail = self.generate_media_page_html(url, uploaded_media, textual_output, thumbnail=thumbnail)
-        # # if multiple wall/photos/videos are present the screenshot will only grab the 1st
-        screenshot = self.get_screenshot(url)
-        wacz = self.get_wacz(url)
-        return ArchiveResult(status="success", cdn_url=page_cdn, screenshot=screenshot, hash=page_hash, thumbnail=thumbnail, thumbnail_index=thumbnail_index, timestamp=datetime, title=title, wacz=wacz)
--- a/archivers/wayback_archiver.py
+++ b/archivers/wayback_archiver.py
@@ -1,89 +0,0 @@
-import time, requests
-
-from loguru import logger
-from bs4 import BeautifulSoup
-
-from storages import Storage
-from .base_archiver import Archiver, ArchiveResult
-from configs import Config
-
-
-class WaybackArchiver(Archiver):
-    """
-    This archiver could implement a check_if_exists by going to "https://web.archive.org/web/{url}"
-    but that might not be desirable since the webpage might have been archived a long time ago and thus have changed
-    """
-    name = "wayback"
-
-    def __init__(self, storage: Storage, config: Config):
-        super(WaybackArchiver, self).__init__(storage, config)
-        self.config = config.wayback_config
-        self.seen_urls = {}
-
-    def download(self, url, check_if_exists=False):
-        if self.config is None:
-            logger.error('Missing Wayback config')
-            return False
-        if check_if_exists:
-            if url in self.seen_urls: return self.seen_urls[url]
-
-        screenshot = self.get_screenshot(url)
-        wacz = self.get_wacz(url)
-
-        logger.debug(f"POSTing {url=} to web.archive.org")
-        ia_headers = {
-            "Accept": "application/json",
-            "Authorization": f"LOW {self.config.key}:{self.config.secret}"
-        }
-        r = requests.post('https://web.archive.org/save/', headers=ia_headers, data={'url': url})
-
-        if r.status_code != 200:
-            logger.warning(f"Internet archive failed with status of {r.status_code}")
-            return ArchiveResult(status="Internet archive failed", screenshot=screenshot, wacz=wacz)
-
-        if 'job_id' not in r.json() and 'message' in r.json():
-            return self.custom_retry(r.json(), screenshot=screenshot, wacz=wacz)
-
-        job_id = r.json()['job_id']
-        logger.debug(f"GETting status for {job_id=} on {url=}")
-        status_r = requests.get(f'https://web.archive.org/save/status/{job_id}', headers=ia_headers)
-        retries = 0
-
-        # TODO: make the job queue parallel -> consider propagation of results back to sheet though
-        # wait 90-120 seconds for the archive job to finish
-        while (status_r.status_code != 200 or status_r.json()['status'] == 'pending') and retries < 30:
-            time.sleep(3)
-            try:
-                logger.debug(f"GETting status for {job_id=} on {url=} [{retries=}]")
-                status_r = requests.get(f'https://web.archive.org/save/status/{job_id}', headers=ia_headers)
-            except:
-                time.sleep(1)
-            retries += 1
-
-        if status_r.status_code != 200:
-            return ArchiveResult(status=f"Internet archive failed: check https://web.archive.org/save/status/{job_id}", screenshot=screenshot, wacz=wacz)
-
-        status_json = status_r.json()
-        if status_json['status'] != 'success':
-            return self.custom_retry(status_json, screenshot=screenshot, wacz=wacz)
-
-        archive_url = f"https://web.archive.org/web/{status_json['timestamp']}/{status_json['original_url']}"
-
-        try:
-            req = requests.get(archive_url)
-            parsed = BeautifulSoup(req.content, 'html.parser')
-            title = parsed.find_all('title')[0].text
-            if title == 'Wayback Machine':
-                title = 'Could not get title'
-        except:
-            title = "Could not get title"
-        self.seen_urls[url] = ArchiveResult(status='success', cdn_url=archive_url, title=title, screenshot=screenshot, wacz=wacz)
-        return self.seen_urls[url]
-
-    def custom_retry(self, json_data, **kwargs):
-        logger.warning(f"Internet archive failed json \n {json_data}")
-        if "please try again" in str(json_data).lower():
-            return self.signal_retry_in(**kwargs)
-        if "this host has been already captured" in str(json_data).lower():
-            return self.signal_retry_in(**kwargs, min_seconds=86400, max_seconds=129600)  # 24h to 36h later
-        return ArchiveResult(status=f"Internet archive failed: {json_data}", **kwargs)
--- a/archivers/youtubedl_archiver.py
+++ b/archivers/youtubedl_archiver.py
@@ -1,118 +0,0 @@
-
-import os, datetime
-
-import yt_dlp
-from loguru import logger
-
-from .base_archiver import Archiver, ArchiveResult
-from storages import Storage
-from configs import Config
-
-
-class YoutubeDLArchiver(Archiver):
-    name = "youtube_dl"
-    ydl_opts = {'outtmpl': f'{Storage.TMP_FOLDER}%(id)s.%(ext)s', 'quiet': False}
-
-    def __init__(self, storage: Storage, config: Config):
-        super().__init__(storage, config)
-        self.fb_cookie = config.facebook_cookie
-
-    def download(self, url, check_if_exists=False):
-        netloc = self.get_netloc(url)
-        if netloc in ['facebook.com', 'www.facebook.com'] and self.fb_cookie:
-            logger.debug('Using Facebook cookie')
-            yt_dlp.utils.std_headers['cookie'] = self.fb_cookie
-
-        ydl = yt_dlp.YoutubeDL(YoutubeDLArchiver.ydl_opts)
-        cdn_url = None
-        status = 'success'
-
-        try:
-            info = ydl.extract_info(url, download=False)
-        except yt_dlp.utils.DownloadError as e:
-            logger.debug(f'No video - Youtube normal control flow: {e}')
-            return False
-        except Exception as e:
-            logger.debug(f'ytdlp exception which is normal for example a facebook page with images only will cause a IndexError: list index out of range. Exception here is: \n  {e}')
-            return False
-
-        if info.get('is_live', False):
-            logger.warning("Live streaming media, not archiving now")
-            return ArchiveResult(status="Streaming media")
-
-        if 'twitter.com' in netloc:
-            if 'https://twitter.com/' in info['webpage_url']:
-                logger.info('Found https://twitter.com/ in the download url from Twitter')
-            else:
-                logger.info('Found a linked video probably in a link in a tweet - not getting that video as there may be images in the tweet')
-                return False
-
-        if check_if_exists:
-            if 'entries' in info:
-                if len(info['entries']) > 1:
-                    logger.warning('YoutubeDLArchiver succeeded but cannot archive channels or pages with multiple videos')
-                    return False
-                elif len(info['entries']) == 0:
-                    logger.warning(
-                        'YoutubeDLArchiver succeeded but did not find video')
-                    return False
-
-                filename = ydl.prepare_filename(info['entries'][0])
-            else:
-                filename = ydl.prepare_filename(info)
-
-            key = self.get_key(filename)
-
-            if self.storage.exists(key):
-                status = 'already archived'
-                cdn_url = self.storage.get_cdn_url(key)
-
-        # sometimes this results in a different filename, so do this again
-        info = ydl.extract_info(url, download=True)
-
-        # TODO: add support for multiple videos
-        if 'entries' in info:
-            if len(info['entries']) > 1:
-                logger.warning(
-                    'YoutubeDLArchiver cannot archive channels or pages with multiple videos')
-                return False
-            else:
-                info = info['entries'][0]
-
-        filename = ydl.prepare_filename(info)
-
-        if not os.path.exists(filename):
-            filename = filename.split('.')[0] + '.mkv'
-
-        if status != 'already archived':
-            key = self.get_key(filename)
-            self.storage.upload(filename, key)
-
-            # filename ='tmp/sDE-qZdi8p8.webm'
-            # key ='SM0022/youtube_dl_sDE-qZdi8p8.webm'
-            cdn_url = self.storage.get_cdn_url(key)
-
-        hash = self.get_hash(filename)
-        screenshot = self.get_screenshot(url)
-        wacz = self.get_wacz(url)
-
-        # get duration
-        duration = info.get('duration')
-
-        # get thumbnails
-        try:
-            key_thumb, thumb_index = self.get_thumbnails(filename, key, duration=duration)
-        except:
-            key_thumb = ''
-            thumb_index = 'Could not generate thumbnails'
-
-        os.remove(filename)
-
-        timestamp = None
-        if 'timestamp' in info and info['timestamp'] is not None:
-            timestamp = datetime.datetime.utcfromtimestamp(info['timestamp']).replace(tzinfo=datetime.timezone.utc).isoformat()
-        elif 'upload_date' in info and info['upload_date'] is not None:
-            timestamp = datetime.datetime.strptime(info['upload_date'], '%Y%m%d').replace(tzinfo=datetime.timezone.utc)
-
-        return ArchiveResult(status=status, cdn_url=cdn_url, thumbnail=key_thumb, thumbnail_index=thumb_index, duration=duration,
-                             title=info['title'] if 'title' in info else None, timestamp=timestamp, hash=hash, screenshot=screenshot, wacz=wacz)
--- a/auto_archive.py
+++ b/auto_archive.py
@@ -1,170 +0,0 @@
-import os, datetime, traceback, random, tempfile
-
-from loguru import logger
-from slugify import slugify
-from urllib.parse import quote
-
-from archivers import TelethonArchiver, TelegramArchiver, TiktokArchiver, YoutubeDLArchiver, TwitterArchiver, TwitterApiArchiver, VkArchiver, WaybackArchiver, ArchiveResult, Archiver
-from utils import GWorksheet, mkdir_if_not_exists, expand_url
-from configs import Config
-from storages import Storage
-
-random.seed()
-
-
-def update_sheet(gw, row, url, result: ArchiveResult):
-    cell_updates = []
-    row_values = gw.get_row(row)
-
-    def batch_if_valid(col, val, final_value=None):
-        final_value = final_value or val
-        if val and gw.col_exists(col) and gw.get_cell(row_values, col) == '':
-            cell_updates.append((row, col, final_value))
-
-    cell_updates.append((row, 'status', result.status))
-
-    batch_if_valid('archive', result.cdn_url)
-    batch_if_valid('date', True, datetime.datetime.utcnow().replace(tzinfo=datetime.timezone.utc).isoformat())
-    batch_if_valid('thumbnail', result.thumbnail, f'=IMAGE("{result.thumbnail}")')
-    batch_if_valid('thumbnail_index', result.thumbnail_index)
-    batch_if_valid('title', result.title)
-    batch_if_valid('duration', result.duration, str(result.duration))
-    batch_if_valid('screenshot', result.screenshot)
-    batch_if_valid('hash', result.hash)
-    if result.wacz is not None:
-        batch_if_valid('wacz', result.wacz)
-        batch_if_valid('replaywebpage', f'https://replayweb.page/?source={quote(result.wacz)}#view=pages&url={quote(url)}')
-
-    if result.timestamp is not None:
-        if type(result.timestamp) == int:
-            timestamp_string = datetime.datetime.fromtimestamp(result.timestamp).replace(tzinfo=datetime.timezone.utc).isoformat()
-        elif type(result.timestamp) == str:
-            timestamp_string = result.timestamp
-        else:
-            timestamp_string = result.timestamp.isoformat()
-
-        batch_if_valid('timestamp', timestamp_string)
-
-    gw.batch_set_cell(cell_updates)
-
-
-def missing_required_columns(gw: GWorksheet):
-    missing = False
-    for required_col in ['url', 'status']:
-        if not gw.col_exists(required_col):
-            logger.warning(f'Required column for {required_col}: "{gw.columns[required_col]}" not found, skipping worksheet {gw.wks.title}')
-            missing = True
-    return missing
-
-
-def should_process_sheet(c, sheet_name):
-    if len(c.worksheet_allow) and sheet_name not in c.worksheet_allow:
-        # ALLOW rules exist AND sheet name not explicitly allowed
-        return False
-    if len(c.worksheet_block) and sheet_name in c.worksheet_block:
-        # BLOCK rules exist AND sheet name is blocked
-        return False
-    return True
-
-
-def process_sheet(c: Config):
-    sh = c.gsheets_client.open(c.sheet)
-
-    # loop through worksheets to check
-    for ii, wks in enumerate(sh.worksheets()):
-        if not should_process_sheet(c, wks.title):
-            logger.info(f'Ignoring worksheet "{wks.title}" due to allow/block configurations')
-            continue
-
-        logger.info(f'Opening worksheet {ii=}: {wks.title=} {c.header=}')
-        gw = GWorksheet(wks, header_row=c.header, columns=c.column_names)
-
-        if missing_required_columns(gw): continue
-
-        # archives will default to being in a folder 'doc_name/worksheet_name'
-        default_folder = os.path.join(slugify(c.sheet), slugify(wks.title))
-        c.set_folder(default_folder)
-        storage = c.get_storage()
-
-        # loop through rows in worksheet
-        for row in range(1 + c.header, gw.count_rows() + 1):
-            url = gw.get_cell(row, 'url')
-            original_status = gw.get_cell(row, 'status')
-            status = gw.get_cell(row, 'status', fresh=original_status in ['', None] and url != '')
-
-            is_retry = False
-            if url == '' or status not in ['', None]:
-                is_retry = Archiver.should_retry_from_status(status)
-                if not is_retry: continue
-
-            # All checks done - archival process starts here
-            try:
-                gw.set_cell(row, 'status', 'Archive in progress')
-                url = expand_url(url)
-                c.set_folder(gw.get_cell_or_default(row, 'folder', default_folder, when_empty_use_default=True))
-
-                # make a new driver so each spreadsheet row is idempotent
-                c.recreate_webdriver()
-
-                # order matters, first to succeed excludes remaining
-                active_archivers = [
-                    TelethonArchiver(storage, c),
-                    TiktokArchiver(storage, c),
-                    TwitterApiArchiver(storage, c),
-                    YoutubeDLArchiver(storage, c),
-                    TelegramArchiver(storage, c),
-                    TwitterArchiver(storage, c),
-                    VkArchiver(storage, c),
-                    WaybackArchiver(storage, c)
-                ]
-
-                for archiver in active_archivers:
-                    logger.debug(f'Trying {archiver} on {row=}')
-
-                    try:
-                        result = archiver.download(url, check_if_exists=c.check_if_exists)
-                    except KeyboardInterrupt as e: raise e  # so the higher level catch can catch it
-                    except Exception as e:
-                        result = False
-                        logger.error(f'Got unexpected error in row {row} with {archiver.name} for {url=}: {e}\n{traceback.format_exc()}')
-
-                    if result:
-                        success = result.status in ['success', 'already archived']
-                        result.status = f"{archiver.name}: {result.status}"
-                        if success:
-                            logger.success(f'{archiver.name} succeeded on {row=}, {url=}')
-                            break
-                        # only 1 retry possible for now
-                        if is_retry and Archiver.is_retry(result.status):
-                            result.status = Archiver.remove_retry(result.status)
-                        logger.warning(f'{archiver.name} did not succeed on {row=}, final status: {result.status}')
-
-                if result:
-                    update_sheet(gw, row, url, result)
-                else:
-                    gw.set_cell(row, 'status', 'failed: no archiver')
-            except KeyboardInterrupt:
-                # catches keyboard interruptions to do a clean exit
-                logger.warning(f"caught interrupt on {row=}, {url=}")
-                gw.set_cell(row, 'status', '')
-                c.destroy_webdriver()
-                exit()
-            except Exception as e:
-                logger.error(f'Got unexpected error in row {row} for {url=}: {e}\n{traceback.format_exc()}')
-                gw.set_cell(row, 'status', 'failed: unexpected error (see logs)')
-        logger.success(f'Finished worksheet {wks.title}')
-
-
-@logger.catch
-def main():
-    c = Config()
-    c.parse()
-    logger.info(f'Opening document {c.sheet} for header {c.header}')
-    with tempfile.TemporaryDirectory(dir="./") as tmpdir:
-        Storage.TMP_FOLDER = tmpdir
-        process_sheet(c)
-        c.destroy_webdriver()
-
-
-if __name__ == '__main__':
-    main()
--- a/auto_auto_archive.py
+++ b/auto_auto_archive.py
@@ -1,29 +0,0 @@
-import tempfile
-import auto_archive
-from loguru import logger
-from configs import Config
-from storages import Storage
-
-
-def main():
-    c = Config()
-    c.parse()
-    logger.info(f'Opening document {c.sheet} to look for sheet names to archive')
-
-    gc = c.gsheets_client
-    sh = gc.open(c.sheet)
-
-    wks = sh.get_worksheet(0)
-    values = wks.get_all_values()
-
-    with tempfile.TemporaryDirectory(dir="./") as tmpdir:
-        Storage.TMP_FOLDER = tmpdir
-        for i in range(11, len(values)):
-            c.sheet = values[i][0]
-            logger.info(f"Processing {c.sheet}")
-            auto_archive.process_sheet(c)
-        c.destroy_webdriver()
-
-
-if __name__ == "__main__":
-    main()
--- a/configs/init.py
+++ b/configs/init.py
@@ -1,6 +0,0 @@
-from .config import Config
-from .selenium_config import SeleniumConfig
-from .telethon_config import TelethonConfig
-from .wayback_config import WaybackConfig
-from .twitter_api_config import TwitterApiConfig
-from .vk_config import VkConfig
--- a/configs/browsertrix_config.py
+++ b/configs/browsertrix_config.py
@@ -1,7 +0,0 @@
-from dataclasses import dataclass
-
-@dataclass
-class BrowsertrixConfig:
-    enabled: bool
-    profile: str
-    timeout_seconds: str
--- a/configs/config.py
+++ b/configs/config.py
@@ -1,291 +0,0 @@
-import argparse, yaml, json, os
-import gspread
-from loguru import logger
-from selenium import webdriver
-from dataclasses import asdict
-from selenium.common.exceptions import TimeoutException
-
-from utils import GWorksheet, getattr_or
-from .wayback_config import WaybackConfig
-from .telethon_config import TelethonConfig
-from .selenium_config import SeleniumConfig
-from .vk_config import VkConfig
-from .twitter_api_config import TwitterApiConfig
-from .browsertrix_config import BrowsertrixConfig
-from storages import S3Config, S3Storage, GDStorage, GDConfig, LocalStorage, LocalConfig
-
-
-class Config:
-    """
-    Controls the current execution parameters and manages API configurations
-    Usage:
-      c = Config() # initializes the argument parser
-      c.parse() # parses the values and initializes the Services and API clients
-      # you can then access the Services and APIs like 'c.s3_config'
-    All the configurations available as cmd line options, when included, will 
-    override the configurations in the config.yaml file.
-    Configurations are split between:
-    1. "secrets" containing API keys for generating services - not kept in memory
-    2. "execution" containing specific execution configurations
-    """
-    AVAILABLE_STORAGES = {"s3", "gd", "local"}
-
-    def __init__(self):
-        self.parser = self.get_argument_parser()
-        self.folder = ""
-
-    def parse(self):
-        self.args = self.parser.parse_args()
-        logger.success(f'Command line arguments parsed successfully')
-        self.config_file = self.args.config
-        self.read_config_yaml()
-        logger.info(f'APIs and Services initialized:\n{self}')
-
-    def read_config_yaml(self):
-        with open(self.config_file, "r", encoding="utf-8") as inf:
-            self.config = yaml.safe_load(inf)
-
-        # ----------------------EXECUTION - execution configurations
-        execution = self.config.get("execution", {})
-
-        self.sheet = getattr_or(self.args, "sheet", execution.get("sheet"))
-        assert self.sheet is not None, "'sheet' must be provided either through command line or configuration file"
-
-        def ensure_set(l):
-            # always returns a set of strings, can receive a set or a string
-            l = l if isinstance(l, list) else [l]
-            return set([x for x in l if isinstance(x, str) and len(x) > 0])
-        self.worksheet_allow = ensure_set(execution.get("worksheet_allow", []))
-        self.worksheet_block = ensure_set(execution.get("worksheet_block", []))
-
-        self.header = int(getattr_or(self.args, "header", execution.get("header", 1)))
-        self.storage = getattr_or(self.args, "storage", execution.get("storage", "s3"))
-        self.save_logs = getattr(self.args, "save_logs") or execution.get("save_logs", False)
-        if self.save_logs:
-            self.set_log_files()
-        self.check_if_exists = getattr(self.args, "check_if_exists") or execution.get("check_if_exists", False)
-
-        # Column names come from config and can be overwritten by CMD
-        # in the end all are considered as lower case
-        config_column_names = execution.get("column_names", {})
-        self.column_names = {}
-        for k in GWorksheet.COLUMN_NAMES.keys():
-            self.column_names[k] = getattr_or(self.args, k, config_column_names.get(k, GWorksheet.COLUMN_NAMES[k])).lower()
-
-        # selenium driver
-        selenium_configs = execution.get("selenium", {})
-        self.selenium_config = SeleniumConfig(
-            timeout_seconds=int(selenium_configs.get("timeout_seconds", SeleniumConfig.timeout_seconds)),
-            window_width=int(selenium_configs.get("window_width", SeleniumConfig.window_width)),
-            window_height=int(selenium_configs.get("window_height", SeleniumConfig.window_height))
-        )
-        self.webdriver = "not initialized"
-
-        # browsertrix config
-        browsertrix_configs = execution.get("browsertrix", {})
-        if len(browsertrix_profile := browsertrix_configs.get("profile", "")):
-            browsertrix_profile = os.path.abspath(browsertrix_profile)
-        self.browsertrix_config = BrowsertrixConfig(
-            enabled=bool(browsertrix_configs.get("enabled", False)),
-            profile=browsertrix_profile,
-            timeout_seconds=browsertrix_configs.get("timeout_seconds", "90")
-        )
-
-        self.hash_algorithm = execution.get("hash_algorithm", "SHA-256")
-
-        # ---------------------- SECRETS - APIs and service configurations
-        secrets = self.config.get("secrets", {})
-
-        # assert selected storage credentials exist
-        for key, name in [("s3", "s3"), ("gd", "google_drive"), ("local", "local")]:
-            assert self.storage != key or name in secrets, f"selected storage '{key}' requires secrets.'{name}' in {self.config_file}"
-
-        # google sheets config
-        self.gsheets_client = gspread.service_account(
-            filename=secrets.get("google_sheets", {}).get("service_account", 'service_account.json')
-        )
-
-        # facebook config
-        self.facebook_cookie = secrets.get("facebook", {}).get("cookie", None)
-
-        # s3 config
-        if "s3" in secrets:
-            s3 = secrets["s3"]
-            self.s3_config = S3Config(
-                bucket=s3["bucket"],
-                region=s3["region"],
-                key=s3["key"],
-                secret=s3["secret"],
-                endpoint_url=s3.get("endpoint_url", S3Config.endpoint_url),
-                cdn_url=s3.get("cdn_url", S3Config.cdn_url),
-                key_path=s3.get("key_path", S3Config.key_path),
-                private=getattr_or(self.args, "s3-private", s3.get("private", S3Config.private))
-            )
-
-        # GDrive config
-        if "google_drive" in secrets:
-            gd = secrets["google_drive"]
-            self.gd_config = GDConfig(
-                root_folder_id=gd.get("root_folder_id"),
-                oauth_token_filename=gd.get("oauth_token_filename"),
-                service_account=gd.get("service_account", GDConfig.service_account)
-            )
-
-        if "local" in secrets:
-            self.local_config = LocalConfig(
-                save_to=secrets["local"].get("save_to", LocalConfig.save_to),
-            )
-
-        # wayback machine config
-        if "wayback" in secrets:
-            self.wayback_config = WaybackConfig(
-                key=secrets["wayback"]["key"],
-                secret=secrets["wayback"]["secret"],
-            )
-        else:
-            self.wayback_config = None
-            logger.debug(f"'wayback' key not present in the {self.config_file=}")
-
-        # telethon config
-        if "telegram" in secrets:
-            self.telegram_config = TelethonConfig(
-                api_id=secrets["telegram"]["api_id"],
-                api_hash=secrets["telegram"]["api_hash"],
-                bot_token=secrets["telegram"].get("bot_token", None)
-            )
-        else:
-            self.telegram_config = None
-            logger.debug(f"'telegram' key not present in the {self.config_file=}")
-
-        # twitter config
-        if "twitter" in secrets:
-            self.twitter_config = TwitterApiConfig(
-                bearer_token=secrets["twitter"].get("bearer_token"),
-                consumer_key=secrets["twitter"].get("consumer_key"),
-                consumer_secret=secrets["twitter"].get("consumer_secret"),
-                access_token=secrets["twitter"].get("access_token"),
-                access_secret=secrets["twitter"].get("access_secret"),
-            )
-        else:
-            self.twitter_config = None
-            logger.debug(f"'twitter' key not present in the {self.config_file=}")
-
-        # vk config
-        if "vk" in secrets:
-            self.vk_config = VkConfig(
-                username=secrets["vk"]["username"],
-                password=secrets["vk"]["password"]
-            )
-        else:
-            self.vk_config = None
-            logger.debug(f"'vk' key not present in the {self.config_file=}")
-
-        del self.config["secrets"]  # delete to prevent leaks
-
-    def set_log_files(self):
-        # called only when config.execution.save_logs=true
-        logger.add("logs/1trace.log", level="TRACE")
-        logger.add("logs/2info.log", level="INFO")
-        logger.add("logs/3success.log", level="SUCCESS")
-        logger.add("logs/4warning.log", level="WARNING")
-        logger.add("logs/5error.log", level="ERROR")
-
-    def get_argument_parser(self):
-        """
-        Creates the CMD line arguments. 'python auto_archive.py --help'
-        """
-        parser = argparse.ArgumentParser(description='Automatically archive social media posts, videos, and images from a Google Sheets document. The command line arguments will always override the configurations in the provided YAML config file (--config), only some high-level options are allowed via the command line and the YAML configuration file is the preferred method. The sheet must have the "url" and "status" for the archiver to work. ')
-
-        parser.add_argument('--config', action='store', dest='config', help='the filename of the YAML configuration file (defaults to \'config.yaml\')', default='config.yaml')
-        parser.add_argument('--storage', action='store', dest='storage', help='which storage to use [execution.storage in config.yaml]', choices=Config.AVAILABLE_STORAGES)
-        parser.add_argument('--sheet', action='store', dest='sheet', help='the name of the google sheets document [execution.sheet in config.yaml]')
-        parser.add_argument('--header', action='store', dest='header', help='1-based index for the header row [execution.header in config.yaml]')
-        parser.add_argument('--check-if-exists', action='store_true', dest='check_if_exists', help='when possible checks if the URL has been archived before and does not archive the same URL twice [exceution.check_if_exists]')
-        parser.add_argument('--save-logs', action='store_true', dest='save_logs', help='creates or appends execution logs to files logs/LEVEL.log [exceution.save_logs]')
-        parser.add_argument('--s3-private', action='store_true', help='Store content without public access permission (only for storage=s3) [secrets.s3.private in config.yaml]')
-
-        for k, v in GWorksheet.COLUMN_NAMES.items():
-            help = f"the name of the column to FILL WITH {k} (default='{v}')"
-            if k in ["url", "folder"]:
-                help = f"the name of the column to READ {k} FROM (default='{v}')"
-            parser.add_argument(f'--col-{k}', action='store', dest=k, help=help)
-
-        return parser
-
-    def set_folder(self, folder):
-        """
-        update the folder in each of the storages
-        """
-        self.folder = folder
-        logger.info(f"setting folder to {folder}")
-        # s3
-        if hasattr(self, "s3_config"): self.s3_config.folder = folder
-        if hasattr(self, "s3_storage"): self.s3_storage.folder = folder
-        # gdrive
-        if hasattr(self, "gd_config"): self.gd_config.folder = folder
-        if hasattr(self, "gd_storage"): self.gd_storage.folder = folder
-        # local
-        if hasattr(self, "local_config"): self.local_config.folder = folder
-        if hasattr(self, "local_storage"): self.local_storage.folder = folder
-
-    def get_storage(self):
-        """
-        returns the configured type of storage, creating if needed
-        """
-        if self.storage == "s3":
-            self.s3_storage = getattr_or(self, "s3_storage", S3Storage(self.s3_config))
-            return self.s3_storage
-        elif self.storage == "gd":
-            self.gd_storage = getattr_or(self, "gd_storage", GDStorage(self.gd_config))
-            return self.gd_storage
-        elif self.storage == "local":
-            self.local_storage = getattr_or(self, "local_storage", LocalStorage(self.local_config))
-            return self.local_storage
-        raise f"storage {self.storage} not implemented, available: {Config.AVAILABLE_STORAGES}"
-
-    def destroy_webdriver(self):
-        if self.webdriver is not None and type(self.webdriver) != str:
-            self.webdriver.close()
-            self.webdriver.quit()
-            del self.webdriver
-
-    def recreate_webdriver(self):
-        options = webdriver.FirefoxOptions()
-        options.headless = True
-        options.set_preference('network.protocol-handler.external.tg', False)
-        try:
-            new_webdriver = webdriver.Firefox(options=options)
-            # only destroy if creation is successful
-            self.destroy_webdriver()
-            self.webdriver = new_webdriver
-            self.webdriver.set_window_size(self.selenium_config.window_width,
-                                           self.selenium_config.window_height)
-            self.webdriver.set_page_load_timeout(self.selenium_config.timeout_seconds)
-        except TimeoutException as e:
-            logger.error(f"failed to get new webdriver, possibly due to insufficient system resources or timeout settings: {e}")
-
-    def __str__(self) -> str:
-        return json.dumps({
-            "config_file": self.config_file,
-            "sheet": self.sheet,
-            "worksheet_allow": list(self.worksheet_allow),
-            "worksheet_block": list(self.worksheet_block),
-            "storage": self.storage,
-            "header": self.header,
-            "check_if_exists": self.check_if_exists,
-            "hash_algorithm": self.hash_algorithm,
-            "browsertrix_config": asdict(self.browsertrix_config),
-            "save_logs": self.save_logs,
-            "selenium_config": asdict(self.selenium_config),
-            "selenium_webdriver": self.webdriver != None,
-            "s3_config": hasattr(self, "s3_config"),
-            "s3_private": getattr_or(getattr(self, "s3_config", {}), "private", None),
-            "gd_config": hasattr(self, "gd_config"),
-            "local_config": hasattr(self, "local_config"),
-            "wayback_config": self.wayback_config != None,
-            "telegram_config": self.telegram_config != None,
-            "twitter_config": self.twitter_config != None,
-            "vk_config": self.vk_config != None,
-            "gsheets_client": self.gsheets_client != None,
-            "column_names": self.column_names,
-        }, ensure_ascii=False, indent=4)
--- a/configs/selenium_config.py
+++ b/configs/selenium_config.py
@@ -1,8 +0,0 @@
-from dataclasses import dataclass
-
-
-@dataclass
-class SeleniumConfig:
-    timeout_seconds: int = 120
-    window_width: int = 1400
-    window_height: int = 2000
--- a/configs/telethon_config.py
+++ b/configs/telethon_config.py
@@ -1,9 +0,0 @@
-
-from dataclasses import dataclass
-
-
-@dataclass
-class TelethonConfig:
-    api_id: str
-    api_hash: str
-    bot_token: str
--- a/configs/twitter_api_config.py
+++ b/configs/twitter_api_config.py
@@ -1,11 +0,0 @@
-
-from dataclasses import dataclass
-
-
-@dataclass
-class TwitterApiConfig:
-    bearer_token: str
-    consumer_key: str
-    consumer_secret: str
-    access_token: str
-    access_secret: str
--- a/configs/vk_config.py
+++ b/configs/vk_config.py
@@ -1,8 +0,0 @@
-
-from dataclasses import dataclass
-
-
-@dataclass
-class VkConfig:
-    username: str
-    password: str
--- a/configs/wayback_config.py
+++ b/configs/wayback_config.py
@@ -1,8 +0,0 @@
-
-from dataclasses import dataclass
-
-
-@dataclass
-class WaybackConfig:
-    key: str
-    secret: str
--- a/docs/auto-auto.png
+++ b/docs/auto-auto.png
--- a/docs/demo-after.png
+++ b/docs/demo-after.png
--- a/docs/demo-archive.png
+++ b/docs/demo-archive.png
--- a/docs/demo-before.png
+++ b/docs/demo-before.png
--- a/docs/demo-progress.png
+++ b/docs/demo-progress.png
--- a/example.config.yaml
+++ b/example.config.yaml
@@ -1,133 +0,0 @@
---
-secrets:
-  # needed if you use storage=s3
-  s3:
-    # contains S3 info on region, bucket, key and secret
-    region: reg1
-    bucket: my-bucket
-    key: "s3 API key"
-    secret: "s3 API secret"
-    # use region format like such
-    endpoint_url: "https://{region}.digitaloceanspaces.com"
-    # endpoint_url: "https://s3.{region}.amazonaws.com"
-    #use bucket, region, and key (key is the archived file path generated when executing) format like such as:
-    cdn_url: "https://{bucket}.{region}.cdn.digitaloceanspaces.com/{key}"
-    # if private:true S3 urls will not be readable online
-    private: false
-    # with 'random' you can generate a random UUID for the URL instead of a predictable path, useful to still have public but unlisted files, alternative is 'default' or not omitted from config
-    key_path: random
-
-  # needed if you use storage=gd
-  google_drive:
-    # To authenticate with google you have two options (1. service account OR 2. OAuth token)
-
-    # 1. service account - storage space will count towards the developer account
-    # filename can be the same or different file from google_sheets.service_account, defaults to "service_account.json"
-    # service_account: "service_account.json"
-
-    # 2. OAuth token  - storage space will count towards the owner of the GDrive folder
-    # (only 1. or 2. - if both specified then this 2. takes precedence)
-    # needs write access on the server so refresh flow works
-    # To get the token, run the file `create_update_test_oauth_token.py`
-    # you can edit that file if you want a different token filename, default is "gd-token.json"
-    oauth_token_filename: "gd-token.json"
-
-    root_folder_id: copy XXXX from https://drive.google.com/drive/folders/XXXX
-
-  # needed if you use storage=local
-  local:
-    # local path to save files in
-    save_to: "./local_archive"
-
-  wayback:
-    # to get credentials visit https://archive.org/account/s3.php
-    key: your API key
-    secret: your API secret
-
-  telegram:
-    # to get credentials see: https://telegra.ph/How-to-get-Telegram-APP-ID--API-HASH-05-27
-    api_id: your API key, see
-    api_hash: your API hash
-    # optional, but allows access to more content such as large videos, talk to @botfather
-    bot_token: your bot-token
-
-  # twitter configuration - API V2 only
-  # if you don't provide credentials the less-effective unofficial TwitterArchiver will be used instead
-  twitter:
-    # either bearer_token only
-    bearer_token: ""
-    # OR all of the below
-    consumer_key: ""
-    consumer_secret: ""
-    access_token: ""
-    access_secret: ""
-
-  # vkontakte (vk.com) credentials
-  vk:
-    username: "phone number or email"
-    password: "password"
-
-  google_sheets:
-    # local filename: defaults to service_account.json, see https://gspread.readthedocs.io/en/latest/oauth2.html#for-bots-using-service-account
-    service_account: "service_account.json"
-
-  facebook:
-    # optional facebook cookie to have more access to content, from browser, looks like 'cookie: datr= xxxx'
-    cookie: ""
-execution:
-  # can be overwritten with CMD --sheet=
-  sheet: your-sheet-name
-
-  # block or allow worksheets by name, instead of defaulting to checking all worksheets in a Spreadsheet
-  # worksheet_allow and worksheet_block can be single values or lists
-  # if worksheet_allow is specified, worksheet_block is ignored
-  # worksheet_allow:
-  #   - Sheet1
-  #   - "Sheet 2"
-  # worksheet_block: BlockedSheet
-
-  # which row of your tabs contains the header, can be overwritten with CMD --header=
-  header: 1
-  # which storage to use, can be overwritten with CMD --storage=
-  storage: s3
-  # defaults to false, when true will try to avoid duplicate URL archives
-  check_if_exists: true
-
-  # choose a hash algorithm (either SHA-256 or SHA3-512, defaults to SHA-256)
-  # hash_algorithm: SHA-256
-
-  # optional configurations for the selenium browser that takes screenshots, these are the defaults
-  selenium:
-    # values under 10s might mean screenshots fail to grab screenshot
-    timeout_seconds: 120
-    window_width: 1400
-    window_height: 2000
-
-  # optional browsertrix configuration (for profile generation see https://github.com/webrecorder/browsertrix-crawler#creating-and-using-browser-profiles)
-  # browsertrix will capture a WACZ archive of the page which can then be seen as the original on replaywebpage
-  browsertrix:
-    enabled: true # defaults to false
-    profile: "./browsertrix/crawls/profile.tar.gz"
-    timeout_seconds: 120 # defaults to 90s
-  # puts execution logs into /logs folder, defaults to false
-  save_logs: true
-  # custom column names, only needed if different from default, can be overwritten with CMD --col-NAME="VALUE"
-  # url and status are the only columns required to be present in the google sheet
-  column_names:
-    url: link
-    status: archive status
-    archive: archive location
-    # use this column to override default location data
-    folder: folder
-    date: archive date
-    thumbnail: thumbnail
-    thumbnail_index: thumbnail index
-    timestamp: upload timestamp
-    title: upload title
-    duration: duration
-    screenshot: screenshot
-    hash: hash
-    wacz: wacz
-    # if you want the replaypage to work, make sure to allow CORS on your bucket
-    replaywebpage: replaywebpage
-
--- a/example.orchestration.yaml
+++ b/example.orchestration.yaml
@@ -0,0 +1,121 @@
+steps:
+  # only 1 feeder allowed
+  feeder: gsheet_feeder # defaults to cli_feeder
+  archivers: # order matters, uncomment to activate
+    # - vk_archiver
+    # - telethon_archiver
+    # - telegram_archiver
+    # - twitter_archiver
+    # - twitter_api_archiver
+    # - instagram_tbot_archiver
+    # - instagram_archiver
+    # - tiktok_archiver
+    - youtubedl_archiver
+    - wayback_archiver_enricher
+  enrichers:
+    - hash_enricher
+    # - screenshot_enricher
+    # - thumbnail_enricher
+    # - wayback_archiver_enricher
+    # - wacz_enricher
+    
+  formatter: html_formatter # defaults to mute_formatter
+  storages:
+    - local_storage
+    # - s3_storage
+    # - gdrive_storage
+  databases:
+    - console_db
+    # - csv_db
+    # - gsheet_db
+    # - mongo_db
+
+configurations:
+  gsheet_feeder:
+    sheet: "your sheet name"
+    header: 1
+    service_account: "secrets/service_account.json"
+    # allow_worksheets: "only parse this worksheet"
+    # block_worksheets: "blocked sheet 1,blocked sheet 2"
+    use_sheet_names_in_stored_paths: false
+    columns:
+      url: link
+      status: archive status
+      folder: destination folder
+      archive: archive location
+      date: archive date
+      thumbnail: thumbnail
+      timestamp: upload timestamp
+      title: upload title
+      text: textual content
+      screenshot: screenshot
+      hash: hash
+      wacz: wacz
+      replaywebpage: replaywebpage
+  instagram_tbot_archiver:
+    api_id: "TELEGRAM_BOT_API_ID"
+    api_hash: "TELEGRAM_BOT_API_HASH"
+    # session_file: "secrets/anon"
+  telethon_archiver:
+    api_id: "TELEGRAM_BOT_API_ID"
+    api_hash: "TELEGRAM_BOT_API_HASH"
+    # session_file: "secrets/anon"
+    join_channels: false
+    channel_invites: # if you want to archive from private channels
+      - invite: https://t.me/+123456789
+        id: 0000000001
+      - invite: https://t.me/+123456788
+        id: 0000000002
+
+  twitter_api_archiver:
+    # either bearer_token only
+    bearer_token: "TWITTER_BEARER_TOKEN"
+    # OR all of the below
+    # consumer_key: ""
+    # consumer_secret: ""
+    # access_token: ""
+    # access_secret: ""
+  instagram_archiver:
+    username: "INSTAGRAM_USERNAME"
+    password: "INSTAGRAM_PASSWORD"
+    # session_file: "secrets/instaloader.session"
+
+  vk_archiver:
+    username: "or phone number"
+    password: "vk pass"
+    session_file: "secrets/vk_config.v2.json"
+
+  screenshot_enricher:
+    width: 1280
+    height: 2300
+  wayback_archiver_enricher:
+    timeout: 10
+    key: "wayback key"
+    secret: "wayback secret"
+  hash_enricher:
+    algorithm: "SHA3-512" # can also be SHA-256
+  wacz_enricher:
+    profile: secrets/profile.tar.gz
+  local_storage:
+    save_to: "./local_archive"
+    save_absolute: true
+    filename_generator: static
+    path_generator: flat
+  s3_storage:
+    bucket: your-bucket-name
+    region: reg1
+    key: S3_KEY
+    secret: S3_SECRET
+    endpoint_url: "https://{region}.digitaloceanspaces.com"
+    cdn_url: "https://{bucket}.{region}.cdn.digitaloceanspaces.com/{key}"
+    # if private:true S3 urls will not be readable online
+    private: false
+    # with 'random' you can generate a random UUID for the URL instead of a predictable path, useful to still have public but unlisted files, alternative is 'default' or not omitted from config
+    key_path: random
+
+  gdrive_storage:
+    path_generator: url
+    filename_generator: random
+    root_folder_id: folder_id_from_url
+    oauth_token: secrets/gd-token.json # needs to be generated with scripts/create_update_gdrive_oauth_token.py
+    service_account: "secrets/service_account.json"
--- a/pyproject.toml
+++ b/pyproject.toml
@@ -0,0 +1,4 @@
+[build-system]
+requires = ["setuptools", "wheel", "setuptools-pipfile"]
+build-backend = "setuptools.build_meta"
+[tool.setuptools-pipfile]
--- a/scripts/create_update_gdrive_oauth_token.py
+++ b/scripts/create_update_gdrive_oauth_token.py
@@ -1,4 +1,5 @@
 import os.path
+import click, json

 from google.auth.transport.requests import Request
 from google.oauth2.credentials import Credentials
@@ -6,27 +7,42 @@ from google_auth_oauthlib.flow import InstalledAppFlow
 from googleapiclient.discovery import build
 from googleapiclient.errors import HttpError

-# If creating for first time download the OAuth Client Ids json `credentials.json` from https://console.cloud.google.com/apis/credentials OAuth 2.0 Client IDs
-# add "http://localhost:55192/" to the list of "Authorised redirect URIs"
-# https://davemateer.com/2022/04/28/google-drive-with-python for more information
-
 # You can run this code to get a new token and verify it belongs to the correct user
 # This token will be refresh automatically by the auto-archiver
-
 # Code below from https://developers.google.com/drive/api/quickstart/python
+# Example invocation: py scripts/create_update_gdrive_oauth_token.py -c secrets/credentials.json -t secrets/gd-token.json

 SCOPES = ['https://www.googleapis.com/auth/drive']


-def main():
-    token_file = 'gd-token.json'
-    creds = None
-
+@click.command(
+    help="script to generate Google Drive OAuth token to use gdrive_storage, requires credentials.json and outputs gd-token.json, if you don't have credentials.json go to https://console.cloud.google.com/apis/credentials. Be sure to add 'http://localhost:55192/' to the Authorized redirect URIs in your OAuth App. More info:  https://davemateer.com/2022/04/28/google-drive-with-python"
+)
+@click.option(
+    "--credentials",
+    "-c",
+    type=click.Path(exists=True),
+    help="path to the credentials.json file downloaded from https://console.cloud.google.com/apis/credentials",
+    required=True
+)
+@click.option(
+    "--token",
+    "-t",
+    type=click.Path(exists=False),
+    default="gd-token.json",
+    help="file where to place the OAuth token, defaults to gd-token.json which you must then move to where your orchestration file points to, defaults to gd-token.json",
+    required=True
+)
+def main(credentials, token):
    # The file token.json stores the user's access and refresh tokens, and is
-    # created automatically when the authorization flow completes for the first
-    # time.
-    if os.path.exists(token_file):
-        creds = Credentials.from_authorized_user_file(token_file, SCOPES)
+    # created automatically when the authorization flow completes for the first time.
+    creds = None
+    if os.path.exists(token):
+        with open(token, 'r') as stream:
+            creds_json = json.load(stream)
+            # creds = Credentials.from_authorized_user_file(creds_json, SCOPES)
+            creds_json['refresh_token'] = creds_json.get("refresh_token", "")
+            creds = Credentials.from_authorized_user_info(creds_json, SCOPES)

    # If there are no (valid) credentials available, let the user log in.
    if not creds or not creds.valid:
@@ -36,10 +52,10 @@ def main():
        else:
            print('First run through so putting up login dialog')
            # credentials.json downloaded from https://console.cloud.google.com/apis/credentials
-            flow = InstalledAppFlow.from_client_secrets_file('credentials.json', SCOPES)
+            flow = InstalledAppFlow.from_client_secrets_file(credentials, SCOPES)
            creds = flow.run_local_server(port=55192)
        # Save the credentials for the next run
-        with open(token_file, 'w') as token:
+        with open(token, 'w') as token:
            print('Saving new token')
            token.write(creds.to_json())
    else:
--- a/scripts/release.sh
+++ b/scripts/release.sh
@@ -0,0 +1,19 @@
+
+#!/bin/bash
+
+set -e
+
+TAG=$(python -c 'from src.auto_archiver.version import __version__; print("v" + __version__)')
+
+read -p "Creating new release for $TAG. Do you want to continue? [Y/n] " prompt
+
+if [[ $prompt == "y" || $prompt == "Y" || $prompt == "yes" || $prompt == "Yes" ]]; then
+    git add -A
+    git commit -m "Bump version to $TAG for release" || true && git push
+    echo "Creating new git tag $TAG"
+    git tag "$TAG" -m "$TAG"
+    git push --tags
+else
+    echo "Cancelled"
+    exit 1
+fi
--- a/setup.cfg
+++ b/setup.cfg
@@ -0,0 +1,53 @@
+[metadata]
+name = auto_archiver
+version = attr: auto_archiver.version.__version__
+author = Bellingcat
+author_email = tech@bellingcat.com
+description = Easily archive online media content
+long_description = file: README.md
+long_description_content_type = text/markdown
+keywords = archive, oosi, osint, scraping
+license = MIT
+classifiers =
+	Intended Audience :: Developers
+	Intended Audience :: Science/Research
+	License :: OSI Approved :: MIT License
+	Programming Language :: Python :: 3
+project_urls = 
+	Source Code = https://github.com/bellingcat/auto-archiver
+	Bug Tracker = https://github.com/bellingcat/auto-archiver/issues
+	Bellingcat = https://www.bellingcat.com
+platforms = any
+
+[options]
+setup_requires =
+    setuptools-pipfile
+zip_safe = False
+package_dir=
+    =src
+packages=find:
+find_packages=true
+python_requires = >=3.8
+
+[options.package_data]
+* = *.html
+
+[options.entry_points]
+console_scripts =
+    auto-archiver = auto_archiver.__main__:main
+
+# [options.extras_require]
+# pdf = ReportLab>=1.2; RXP
+# rest = docutils>=0.3; pack ==1.1, ==1.3
+
+[options.packages.find]
+where=src
+# include=auto_archiver*
+# exclude =
+#     examples*
+#     .eggs*
+#     build*
+#     secrets*
+#     tmp*
+#     docs*
+#     src.tests*
--- a/setup.py
+++ b/setup.py
@@ -0,0 +1,4 @@
+from setuptools import setup
+
+if __name__ == "__main__":
+    setup()
--- a/src/init.py
+++ b/src/init.py
--- a/src/auto_archiver/init.py
+++ b/src/auto_archiver/init.py
@@ -0,0 +1,7 @@
+from . import archivers, databases, enrichers, feeders, formatters, storages, utils, core
+
+# need to manually specify due to cyclical deps
+from .core.orchestrator import ArchivingOrchestrator
+from .core.config import Config
+# making accessible directly
+from .core.metadata import Metadata
--- a/src/auto_archiver/main.py
+++ b/src/auto_archiver/main.py
@@ -0,0 +1,12 @@
+from . import Config
+from . import ArchivingOrchestrator
+
+def main():
+    config = Config()
+    config.parse()
+    orchestrator = ArchivingOrchestrator(config)
+    orchestrator.feed()
+
+
+if __name__ == "__main__":
+    main()
--- a/src/auto_archiver/archivers/init.py
+++ b/src/auto_archiver/archivers/init.py
@@ -0,0 +1,10 @@
+from .archiver import Archiver
+from .telethon_archiver import TelethonArchiver
+from .twitter_archiver import TwitterArchiver
+from .twitter_api_archiver import TwitterApiArchiver
+from .instagram_archiver import InstagramArchiver
+from .instagram_tbot_archiver import InstagramTbotArchiver
+from .tiktok_archiver import TiktokArchiver
+from .telegram_archiver import TelegramArchiver
+from .vk_archiver import VkArchiver
+from .youtubedl_archiver import YoutubeDLArchiver
--- a/src/auto_archiver/archivers/archiver.py
+++ b/src/auto_archiver/archivers/archiver.py
@@ -0,0 +1,64 @@
+from __future__ import annotations
+from abc import abstractmethod
+from dataclasses import dataclass
+import os
+import mimetypes, requests
+
+from ..core import Metadata, Step, ArchivingContext
+
+
+@dataclass
+class Archiver(Step):
+    name = "archiver"
+
+    def __init__(self, config: dict) -> None:
+        # without this STEP.__init__ is not called
+        super().__init__(config)
+
+    def init(name: str, config: dict) -> Archiver:
+        # only for typing...
+        return Step.init(name, config, Archiver)
+
+    def setup(self) -> None:
+        # used when archivers need to login or do other one-time setup
+        pass
+
+    def sanitize_url(self, url: str) -> str:
+        # used to clean unnecessary URL parameters OR unfurl redirect links
+        return url
+
+    def is_rearchivable(self, url: str) -> bool:
+        # archivers can signal if it does not make sense to rearchive a piece of content
+        # default is rearchiving
+        return True
+
+    def _guess_file_type(self, path: str) -> str:
+        """
+        Receives a URL or filename and returns global mimetype like 'image' or 'video'
+        see https://developer.mozilla.org/en-US/docs/Web/HTTP/Basics_of_HTTP/MIME_types/Common_types
+        """
+        mime = mimetypes.guess_type(path)[0]
+        if mime is not None:
+            return mime.split("/")[0]
+        return ""
+
+    def download_from_url(self, url: str, to_filename: str = None, item: Metadata = None) -> str:
+        """
+        downloads a URL to provided filename, or inferred from URL, returns local filename, if item is present will use its tmp_dir
+        """
+        if not to_filename:
+            to_filename = url.split('/')[-1].split('?')[0]
+            if len(to_filename) > 64:
+                to_filename = to_filename[-64:]
+        if item:
+            to_filename = os.path.join(ArchivingContext.get_tmp_dir(), to_filename)
+        headers = {
+            'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/81.0.4044.138 Safari/537.36'
+        }
+        d = requests.get(url, headers=headers)
+        with open(to_filename, 'wb') as f:
+            f.write(d.content)
+        return to_filename
+
+    @abstractmethod
+    def download(self, item: Metadata) -> Metadata: pass
--- a/src/auto_archiver/archivers/instagram_archiver.py
+++ b/src/auto_archiver/archivers/instagram_archiver.py
@@ -0,0 +1,143 @@
+import re, os, shutil, traceback
+import instaloader  # https://instaloader.github.io/as-module.html
+from loguru import logger
+
+from . import Archiver
+from ..core import Metadata
+from ..core import Media
+
+class InstagramArchiver(Archiver):
+    """
+    Uses Instaloader to download either a post (inc images, videos, text) or as much as possible from a profile (posts, stories, highlights, ...)
+    """
+    name = "instagram_archiver"
+
+    # NB: post regex should be tested before profile
+    # https://regex101.com/r/MGPquX/1
+    post_pattern = re.compile(r"(?:(?:http|https):\/\/)?(?:www.)?(?:instagram.com|instagr.am|instagr.com)\/(?:p|reel)\/(\w+)")
+    # https://regex101.com/r/6Wbsxa/1
+    profile_pattern = re.compile(r"(?:(?:http|https):\/\/)?(?:www.)?(?:instagram.com|instagr.am|instagr.com)\/(\w+)")
+    # TODO: links to stories
+
+    def __init__(self, config: dict) -> None:
+        super().__init__(config)
+        # TODO: refactor how configuration validation is done
+        self.assert_valid_string("username")
+        self.assert_valid_string("password")
+        self.assert_valid_string("download_folder")
+        self.assert_valid_string("session_file")
+        self.insta = instaloader.Instaloader(
+            download_geotags=True, download_comments=True, compress_json=False, dirname_pattern=self.download_folder, filename_pattern="{date_utc}_UTC_{target}__{typename}"
+        )
+        try:
+            self.insta.load_session_from_file(self.username, self.session_file)
+        except Exception as e:
+            logger.error(f"Unable to login from session file: {e}\n{traceback.format_exc()}")
+            try:
+                self.insta.login(self.username, config.instagram_self.password)
+                # TODO: wait for this issue to be fixed https://github.com/instaloader/instaloader/issues/1758
+                self.insta.save_session_to_file(self.session_file)
+            except Exception as e2:
+                logger.error(f"Unable to finish login (retrying from file): {e2}\n{traceback.format_exc()}")
+
+    @staticmethod
+    def configs() -> dict:
+        return {
+            "username": {"default": None, "help": "a valid Instagram username"},
+            "password": {"default": None, "help": "the corresponding Instagram account password"},
+            "download_folder": {"default": "instaloader", "help": "name of a folder to temporarily download content to"},
+            "session_file": {"default": "secrets/instaloader.session", "help": "path to the instagram session which saves session credentials"},
+            #TODO: fine-grain
+            # "download_stories": {"default": True, "help": "if the link is to a user profile: whether to get stories information"},
+        }
+
+    def download(self, item: Metadata) -> Metadata:
+        url = item.get_url()
+
+        # detect URLs that we definitely cannot handle
+        post_matches = self.post_pattern.findall(url)
+        profile_matches = self.profile_pattern.findall(url)
+
+        # return if not a valid instagram link
+        if not len(post_matches) and not len(profile_matches): return
+
+        result = None
+        try:
+            os.makedirs(self.download_folder, exist_ok=True)
+            # process if post
+            if len(post_matches):
+                result = self.download_post(url, post_matches[0])
+            # process if profile
+            elif len(profile_matches):
+                result = self.download_profile(url, profile_matches[0])
+        except Exception as e:
+            logger.error(f"Failed to download with instagram archiver due to: {e}, make sure your account credentials are valid.")
+        finally:
+            shutil.rmtree(self.download_folder, ignore_errors=True)
+        return result
+
+    def download_post(self, url: str, post_id: str) -> Metadata:
+        logger.debug(f"Instagram {post_id=} detected in {url=}")
+
+        post = instaloader.Post.from_shortcode(self.insta.context, post_id)
+        if self.insta.download_post(post, target=post.owner_username):
+            return self.process_downloads(url, post.title, post._asdict(), post.date)
+
+    def download_profile(self, url: str, username: str) -> Metadata:
+        # gets posts, posts where username is tagged, igtv postss, stories, and highlights
+        logger.debug(f"Instagram {username=} detected in {url=}")
+
+        profile = instaloader.Profile.from_username(self.insta.context, username)
+        try:
+            for post in profile.get_posts():
+                try: self.insta.download_post(post, target=f"profile_post_{post.owner_username}")
+                except Exception as e: logger.error(f"Failed to download post: {post.shortcode}: {e}")
+        except Exception as e: logger.error(f"Failed profile.get_posts: {e}")
+
+        try:
+            for post in profile.get_tagged_posts():
+                try: self.insta.download_post(post, target=f"tagged_post_{post.owner_username}")
+                except Exception as e: logger.error(f"Failed to download tagged post: {post.shortcode}: {e}")
+        except Exception as e: logger.error(f"Failed profile.get_tagged_posts: {e}")
+
+        try:
+            for post in profile.get_igtv_posts():
+                try: self.insta.download_post(post, target=f"igtv_post_{post.owner_username}")
+                except Exception as e: logger.error(f"Failed to download igtv post: {post.shortcode}: {e}")
+        except Exception as e: logger.error(f"Failed profile.get_igtv_posts: {e}")
+
+        try:
+            for story in self.insta.get_stories([profile.userid]):
+                for item in story.get_items():
+                    try: self.insta.download_storyitem(item, target=f"story_item_{story.owner_username}")
+                    except Exception as e: logger.error(f"Failed to download story item: {item}: {e}")
+        except Exception as e: logger.error(f"Failed get_stories: {e}")
+
+        try:
+            for highlight in self.insta.get_highlights(profile.userid):
+                for item in highlight.get_items():
+                    try: self.insta.download_storyitem(item, target=f"highlight_item_{highlight.owner_username}")
+                    except Exception as e: logger.error(f"Failed to download highlight item: {item}: {e}")
+        except Exception as e: logger.error(f"Failed get_highlights: {e}")
+
+        return self.process_downloads(url, f"@{username}", profile._asdict(), None)
+
+    def process_downloads(self, url, title, content, date):
+        result = Metadata()
+        result.set_title(title).set_content(str(content)).set_timestamp(date)
+
+        try:
+            all_media = []
+            for f in os.listdir(self.download_folder):
+                if os.path.isfile((filename := os.path.join(self.download_folder, f))):
+                    if filename[-4:] == ".txt": continue
+                    all_media.append(Media(filename))
+
+            assert len(all_media) > 1, "No uploaded media found"
+            all_media.sort(key=lambda m: m.filename, reverse=True)
+            for m in all_media:
+                result.add_media(m)
+
+            return result.success("instagram")
+        except Exception as e:
+            logger.error(f"Could not fetch instagram post {url} due to: {e}")
--- a/src/auto_archiver/archivers/instagram_tbot_archiver.py
+++ b/src/auto_archiver/archivers/instagram_tbot_archiver.py
@@ -0,0 +1,77 @@
+
+from telethon.sync import TelegramClient
+from loguru import logger
+import time, os
+from sqlite3 import OperationalError
+from . import Archiver
+from ..core import Metadata, Media, ArchivingContext
+
+
+class InstagramTbotArchiver(Archiver):
+    """
+    calls a telegram bot to fetch instagram posts/stories... and gets available media from it
+    https://github.com/adw0rd/instagrapi
+    https://t.me/instagram_load_bot
+    """
+    name = "instagram_tbot_archiver"
+
+    def __init__(self, config: dict) -> None:
+        super().__init__(config)
+        self.assert_valid_string("api_id")
+        self.assert_valid_string("api_hash")
+        self.timeout = int(self.timeout)
+        try:
+            self.client = TelegramClient(self.session_file, self.api_id, self.api_hash)
+        except OperationalError as e:
+            logger.error(f"Unable to access the {self.session_file} session, please make sure you don't use the same session file here and in telethon_archiver. if you do then disable at least one of the archivers for the 1st time you setup telethon session: {e}")
+
+    @staticmethod
+    def configs() -> dict:
+        return {
+            "api_id": {"default": None, "help": "telegram API_ID value, go to https://my.telegram.org/apps"},
+            "api_hash": {"default": None, "help": "telegram API_HASH value, go to https://my.telegram.org/apps"},
+            "session_file": {"default": "secrets/anon-insta", "help": "optional, records the telegram login session for future usage, '.session' will be appended to the provided value."},
+            "timeout": {"default": 45, "help": "timeout to fetch the instagram content in seconds."},
+        }
+
+    def setup(self) -> None:
+        logger.info(f"SETUP {self.name} checking login...")
+        with self.client.start():
+            logger.success(f"SETUP {self.name} login works.")
+
+    def download(self, item: Metadata) -> Metadata:
+        url = item.get_url()
+        if not "instagram.com" in url: return False
+
+        result = Metadata()
+        tmp_dir = ArchivingContext.get_tmp_dir()
+        with self.client.start():
+            chat = self.client.get_entity("instagram_load_bot")
+            since_id = self.client.send_message(entity=chat, message=url).id
+
+            attempts = 0
+            seen_media = []
+            message = ""
+            time.sleep(3)
+            # media is added before text by the bot so it can be used as a stop-logic mechanism
+            while attempts < (self.timeout - 3) and (not message or not len(seen_media)):
+                attempts += 1
+                time.sleep(1)
+                for post in self.client.iter_messages(chat, min_id=since_id):
+                    since_id = max(since_id, post.id)
+                    if post.media and post.id not in seen_media:
+                        filename_dest = os.path.join(tmp_dir, f'{chat.id}_{post.id}')
+                        media = self.client.download_media(post.media, filename_dest)
+                        if media: 
+                            result.add_media(Media(media))
+                            seen_media.append(post.id)
+                    if post.message: message += post.message
+
+            if "You must enter a URL to a post" in message: 
+                logger.debug(f"invalid link {url=} for {self.name}: {message}")
+                return False
+                
+            if message:
+                result.set_content(message).set_title(message[:128])
+
+            return result.success("insta-via-bot")
--- a/src/auto_archiver/archivers/telegram_archiver.py
+++ b/src/auto_archiver/archivers/telegram_archiver.py
@@ -0,0 +1,76 @@
+import requests, re, html
+from bs4 import BeautifulSoup
+from loguru import logger
+
+from . import Archiver
+from ..core import Metadata, Media
+
+
+class TelegramArchiver(Archiver):
+    """
+    Archiver for telegram that does not require login, but the telethon_archiver is much more advised, will only return if at least one image or one video is found
+    """
+    name = "telegram_archiver"
+
+    def __init__(self, config: dict) -> None:
+        super().__init__(config)
+
+    @staticmethod
+    def configs() -> dict:
+        return {}
+
+    def is_rearchivable(self, url: str) -> bool:
+        # telegram posts are static
+        return False
+
+    def download(self, item: Metadata) -> Metadata:
+        url = item.get_url()
+        # detect URLs that we definitely cannot handle
+        if 't.me' != item.netloc:
+            return False
+
+        headers = {
+            'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/81.0.4044.138 Safari/537.36'
+        }
+
+        # TODO: check if we can do this more resilient to variable URLs
+        if url[-8:] != "?embed=1":
+            url += "?embed=1"
+
+        t = requests.get(url, headers=headers)
+        s = BeautifulSoup(t.content, 'html.parser')
+
+        result = Metadata()
+        result.set_content(html.escape(str(t.content)))
+        if (timestamp := (s.find_all('time') or [{}])[0].get('datetime')):
+            result.set_timestamp(timestamp)
+
+        video = s.find("video")
+        if video is None:
+            logger.warning("could not find video")
+            image_tags = s.find_all(class_="js-message_photo")
+
+            image_urls = []
+            for im in image_tags:
+                urls = [u.replace("'", "") for u in re.findall(r'url\((.*?)\)', im['style'])]
+                image_urls += urls
+
+            if not len(image_urls): return False
+            for img_url in image_urls:
+                result.add_media(Media(self.download_from_url(img_url)))
+        else:
+            video_url = video.get('src')
+            m_video = Media(self.download_from_url(video_url))
+            # extract duration from HTML
+            try:
+                duration = s.find_all('time')[0].contents[0]
+                if ':' in duration:
+                    duration = float(duration.split(
+                        ':')[0]) * 60 + float(duration.split(':')[1])
+                else:
+                    duration = float(duration)
+                m_video.set("duration", duration)
+            except: pass
+            result.add_media(m_video)
+
+        return result.success("telegram")
--- a/src/auto_archiver/archivers/telethon_archiver.py
+++ b/src/auto_archiver/archivers/telethon_archiver.py
@@ -0,0 +1,173 @@
+
+from telethon.sync import TelegramClient
+from telethon.errors import ChannelInvalidError
+from telethon.tl.functions.messages import ImportChatInviteRequest
+from telethon.errors.rpcerrorlist import UserAlreadyParticipantError, FloodWaitError, InviteRequestSentError, InviteHashExpiredError
+from loguru import logger
+from tqdm import tqdm
+import re, time, json, os
+
+from . import Archiver
+from ..core import Metadata, Media, ArchivingContext
+
+
+class TelethonArchiver(Archiver):
+    name = "telethon_archiver"
+    link_pattern = re.compile(r"https:\/\/t\.me(\/c){0,1}\/(.+)\/(\d+)")
+    invite_pattern = re.compile(r"t.me(\/joinchat){0,1}\/\+?(.+)")
+
+    def __init__(self, config: dict) -> None:
+        super().__init__(config)
+        self.assert_valid_string("api_id")
+        self.assert_valid_string("api_hash")
+
+        self.client = TelegramClient(self.session_file, self.api_id, self.api_hash)
+
+    @staticmethod
+    def configs() -> dict:
+        return {
+            "api_id": {"default": None, "help": "telegram API_ID value, go to https://my.telegram.org/apps"},
+            "api_hash": {"default": None, "help": "telegram API_HASH value, go to https://my.telegram.org/apps"},
+            "bot_token": {"default": None, "help": "optional, but allows access to more content such as large videos, talk to @botfather"},
+            "session_file": {"default": "secrets/anon", "help": "optional, records the telegram login session for future usage, '.session' will be appended to the provided value."},
+            "join_channels": {"default": True, "help": "disables the initial setup with channel_invites config, useful if you have a lot and get stuck"},
+            "channel_invites": {
+                "default": {},
+                "help": "(JSON string) private channel invite links (format: t.me/joinchat/HASH OR t.me/+HASH) and (optional but important to avoid hanging for minutes on startup) channel id (format: CHANNEL_ID taken from a post url like https://t.me/c/CHANNEL_ID/1), the telegram account will join any new channels on setup",
+                "cli_set": lambda cli_val, cur_val: dict(cur_val, **json.loads(cli_val))
+            }
+        }
+
+    def is_rearchivable(self, url: str) -> bool:
+        # telegram posts are static
+        return False
+
+    def setup(self) -> None:
+        """
+        1. trigger login process for telegram or proceed if already saved in a session file
+        2. joins channel_invites where needed
+        """
+        logger.info(f"SETUP {self.name} checking login...")
+        with self.client.start():
+            logger.success(f"SETUP {self.name} login works.")
+
+        if self.join_channels and len(self.channel_invites):
+            logger.info(f"SETUP {self.name} joining channels...")
+            with self.client.start():
+                # get currently joined channels
+                # https://docs.telethon.dev/en/stable/modules/custom.html#module-telethon.tl.custom.dialog
+                joined_channel_ids = [c.id for c in self.client.get_dialogs() if c.is_channel]
+                logger.info(f"already part of {len(joined_channel_ids)} channels")
+
+                i = 0
+                pbar = tqdm(desc=f"joining {len(self.channel_invites)} invite links", total=len(self.channel_invites))
+                while i < len(self.channel_invites):
+                    channel_invite = self.channel_invites[i]
+                    channel_id = channel_invite.get("id", False)
+                    invite = channel_invite["invite"]
+                    if (match := self.invite_pattern.search(invite)):
+                        try:
+                            if channel_id:
+                                ent = self.client.get_entity(int(channel_id))  # fails if not a member
+                            else:
+                                ent = self.client.get_entity(invite)  # fails if not a member
+                                logger.warning(f"please add the property id='{ent.id}' to the 'channel_invites' configuration where {invite=}, not doing so can lead to a minutes-long setup time due to telegram's rate limiting.")
+                        except ValueError as e:
+                            logger.info(f"joining new channel {invite=}")
+                            try:
+                                self.client(ImportChatInviteRequest(match.group(2)))
+                            except UserAlreadyParticipantError as e:
+                                logger.info(f"already joined {invite=}")
+                            except InviteRequestSentError:
+                                logger.warning(f"already sent a join request with {invite} still no answer")
+                            except InviteHashExpiredError:
+                                logger.warning(f"{invite=} has expired please find a more recent one")
+                            except Exception as e:
+                                logger.error(f"could not join channel with {invite=} due to {e}")
+                        except FloodWaitError as e:
+                            logger.warning(f"got a flood error, need to wait {e.seconds} seconds")
+                            time.sleep(e.seconds)
+                            continue
+                    else:
+                        logger.warning(f"Invalid invite link {invite}")
+                    i += 1
+                    pbar.update()
+
+    def download(self, item: Metadata) -> Metadata:
+        """
+        if this url is archivable will download post info and look for other posts from the same group with media.
+        can handle private/public channels
+        """
+        url = item.get_url()
+        # detect URLs that we definitely cannot handle
+        match = self.link_pattern.search(url)
+        logger.debug(f"TELETHON: {match=}")
+        if not match: return False
+
+        is_private = match.group(1) == "/c"
+        chat = int(match.group(2)) if is_private else match.group(2)
+        post_id = int(match.group(3))
+
+        result = Metadata()
+
+        # NB: not using bot_token since then private channels cannot be archived: self.client.start(bot_token=self.bot_token)
+        with self.client.start():
+        # with self.client.start(bot_token=self.bot_token):
+            try:
+                post = self.client.get_messages(chat, ids=post_id)
+            except ValueError as e:
+                logger.error(f"Could not fetch telegram {url} possibly it's private: {e}")
+                return False
+            except ChannelInvalidError as e:
+                logger.error(f"Could not fetch telegram {url}. This error may be fixed if you setup a bot_token in addition to api_id and api_hash (but then private channels will not be archived, we need to update this logic to handle both): {e}")
+                return False
+
+            logger.debug(f"TELETHON GOT POST {post=}")
+            if post is None: return False
+
+            media_posts = self._get_media_posts_in_group(chat, post)
+            logger.debug(f'got {len(media_posts)=} for {url=}')
+
+            tmp_dir = ArchivingContext.get_tmp_dir()
+
+            group_id = post.grouped_id if post.grouped_id is not None else post.id
+            title = post.message
+            for mp in media_posts:
+                if len(mp.message) > len(title): title = mp.message  # save the longest text found (usually only 1)
+
+                # media can also be in entities
+                if mp.entities:
+                    other_media_urls = [e.url for e in mp.entities if hasattr(e, "url") and e.url and self._guess_file_type(e.url) in ["video", "image", "audio"]]
+                    if len(other_media_urls):
+                        logger.debug(f"Got {len(other_media_urls)} other media urls from {mp.id=}: {other_media_urls}")
+                    for i, om_url in enumerate(other_media_urls):
+                        filename = self.download_from_url(om_url, f'{chat}_{group_id}_{i}', item)
+                        result.add_media(Media(filename=filename), id=f"{group_id}_{i}")
+
+                filename_dest = os.path.join(tmp_dir, f'{chat}_{group_id}', str(mp.id))
+                filename = self.client.download_media(mp.media, filename_dest)
+                if not filename:
+                    logger.debug(f"Empty media found, skipping {str(mp)=}")
+                    continue
+                result.add_media(Media(filename))
+
+            result.set_content(str(post)).set_title(title).set_timestamp(post.date)
+        return result.success("telethon")
+
+    def _get_media_posts_in_group(self, chat, original_post, max_amp=10):
+        """
+        Searches for Telegram posts that are part of the same group of uploads
+        The search is conducted around the id of the original post with an amplitude
+        of `max_amp` both ways
+        Returns a list of [post] where each post has media and is in the same grouped_id
+        """
+        if getattr(original_post, "grouped_id", None) is None:
+            return [original_post] if getattr(original_post, "media", False) else []
+
+        search_ids = [i for i in range(original_post.id - max_amp, original_post.id + max_amp + 1)]
+        posts = self.client.get_messages(chat, ids=search_ids)
+        media = []
+        for post in posts:
+            if post is not None and post.grouped_id == original_post.grouped_id and post.media is not None:
+                media.append(post)
+        return media
--- a/src/auto_archiver/archivers/tiktok_archiver.py
+++ b/src/auto_archiver/archivers/tiktok_archiver.py
@@ -0,0 +1,58 @@
+import json, os, traceback, uuid
+import tiktok_downloader
+from loguru import logger
+
+from . import Archiver
+from ..core import Metadata, Media, ArchivingContext
+
+
+class TiktokArchiver(Archiver):
+    name = "tiktok_archiver"
+
+    def __init__(self, config: dict) -> None:
+        super().__init__(config)
+
+    @staticmethod
+    def configs() -> dict:
+        return {}
+
+    def is_rearchivable(self, url: str) -> bool:
+        # TikTok posts are static
+        return False
+
+    def download(self, item: Metadata) -> Metadata:
+        url = item.get_url()
+        if 'tiktok.com' not in url:
+            return False
+
+        result = Metadata()
+        try:
+            info = tiktok_downloader.info_post(url)
+            result.set_title(info.desc)
+            result.set_timestamp(info.create_time)
+            result.set_content(json.dumps({
+                "cover": info.cover,
+                "author": info.author,
+                "music_title": info.author,
+                "caption": getattr(info, "caption", info.desc),
+            }, ensure_ascii=False, indent=4))
+        except:
+            error = traceback.format_exc()
+            logger.warning(f'Other Tiktok error {error}')
+
+        try:
+            filename = os.path.join(ArchivingContext.get_tmp_dir(), f'{str(uuid.uuid4())[0:8]}.mp4')
+            tiktok_media = tiktok_downloader.snaptik(url).get_media()
+
+            if len(tiktok_media) <= 0:
+                logger.debug(f"TikTok: could not get media from {url=}")
+                return False
+
+            logger.info(f'downloading video {filename=}')
+            tiktok_media[0].download(filename)
+
+            result.add_media(Media(filename))
+            return result.success("tiktok")
+        except:
+            error = traceback.format_exc()
+            logger.warning(f'Other Tiktok error {error}')
--- a/src/auto_archiver/archivers/twitter_api_archiver.py
+++ b/src/auto_archiver/archivers/twitter_api_archiver.py
@@ -0,0 +1,98 @@
+
+import json, mimetypes
+from datetime import datetime
+from loguru import logger
+from pytwitter import Api
+from slugify import slugify
+
+from . import Archiver
+from .twitter_archiver import TwitterArchiver
+from ..core import Metadata,Media
+
+
+class TwitterApiArchiver(TwitterArchiver, Archiver):
+    name = "twitter_api_archiver"
+
+    def __init__(self, config: dict) -> None:
+        super().__init__(config)
+
+        if self.bearer_token:
+            self.assert_valid_string("bearer_token")
+            self.api = Api(bearer_token=self.bearer_token)
+        elif self.consumer_key and self.consumer_secret and self.access_token and self.access_secret:
+            self.assert_valid_string("consumer_key")
+            self.assert_valid_string("consumer_secret")
+            self.assert_valid_string("access_token")
+            self.assert_valid_string("access_secret")
+            self.api = Api(
+                consumer_key=self.consumer_key, consumer_secret=self.consumer_secret, access_token=self.access_token, access_secret=self.access_secret)
+        assert hasattr(self, "api") and self.api is not None, "Missing Twitter API configurations, please provide either bearer_token OR (consumer_key, consumer_secret, access_token, access_secret) to use this archiver."
+
+    @staticmethod
+    def configs() -> dict:
+        return {
+            "bearer_token": {"default": None, "help": "twitter API bearer_token which is enough for archiving, if not provided you will need consumer_key, consumer_secret, access_token, access_secret"},
+            "consumer_key": {"default": None, "help": "twitter API consumer_key"},
+            "consumer_secret": {"default": None, "help": "twitter API consumer_secret"},
+            "access_token": {"default": None, "help": "twitter API access_token"},
+            "access_secret": {"default": None, "help": "twitter API access_secret"},
+        }
+
+    def download(self, item: Metadata) -> Metadata:
+        url = item.get_url()
+        # detect URLs that we definitely cannot handle
+        username, tweet_id = self.get_username_tweet_id(url)
+        if not username: return False
+
+        try:
+            tweet = self.api.get_tweet(tweet_id, expansions=["attachments.media_keys"], media_fields=["type", "duration_ms", "url", "variants"], tweet_fields=["attachments", "author_id", "created_at", "entities", "id", "text", "possibly_sensitive"])
+        except Exception as e:
+            logger.error(f"Could not get tweet: {e}")
+            return False
+
+        result = Metadata()
+        result.set_title(tweet.data.text)
+        result.set_timestamp(datetime.strptime(tweet.data.created_at, "%Y-%m-%dT%H:%M:%S.%fZ"))
+
+        urls = []
+        if tweet.includes:
+            for i, m in enumerate(tweet.includes.media):
+                media = Media(filename="")
+                if m.url and len(m.url):
+                    media.set("src", m.url)
+                    media.set("duration", (m.duration_ms or 1) // 1000)
+                    mimetype = "image/jpeg"
+                elif hasattr(m, "variants"):
+                    variant = self.choose_variant(m.variants)
+                    if not variant: continue
+                    media.set("src", variant.url)
+                    mimetype = variant.content_type
+                else:
+                    continue
+                logger.info(f"Found media {media}")
+                ext = mimetypes.guess_extension(mimetype)
+                media.filename = self.download_from_url(media.get("src"), f'{slugify(url)}_{i}{ext}', item)
+                result.add_media(media)
+
+        result.set_content(json.dumps({
+            "id": tweet.data.id,
+            "text": tweet.data.text,
+            "created_at": tweet.data.created_at,
+            "author_id": tweet.data.author_id,
+            "geo": tweet.data.geo,
+            "lang": tweet.data.lang,
+            "media": urls
+        }, ensure_ascii=False, indent=4))
+        return result.success("twitter")
+
+    def choose_variant(self, variants):
+        # choosing the highest quality possible
+        variant, bit_rate = None, -1
+        for var in variants:
+            if var.content_type == "video/mp4":
+                if var.bit_rate > bit_rate:
+                    bit_rate = var.bit_rate
+                    variant = var
+            else:
+                variant = var if not variant else variant
+        return variant
--- a/src/auto_archiver/archivers/twitter_archiver.py
+++ b/src/auto_archiver/archivers/twitter_archiver.py
@@ -0,0 +1,148 @@
+import re, requests, mimetypes, json
+from datetime import datetime
+from loguru import logger
+from snscrape.modules.twitter import TwitterTweetScraper, Video, Gif, Photo
+from slugify import slugify
+
+from . import Archiver
+from ..core import Metadata, Media
+
+
+class TwitterArchiver(Archiver):
+    """
+    This Twitter Archiver uses unofficial scraping methods.
+    """
+
+    name = "twitter_archiver"
+    link_pattern = re.compile(r"twitter.com\/(?:\#!\/)?(\w+)\/status(?:es)?\/(\d+)")
+    link_clean_pattern = re.compile(r"(.+twitter\.com\/.+\/\d+)(\?)*.*")
+
+    def __init__(self, config: dict) -> None:
+        super().__init__(config)
+
+    @staticmethod
+    def configs() -> dict:
+        return {}
+
+    def sanitize_url(self, url: str) -> str:
+        # expand URL if t.co and clean tracker GET params
+        if 'https://t.co/' in url:
+            try:
+                r = requests.get(url)
+                logger.debug(f'Expanded url {url} to {r.url}')
+                url = r.url
+            except:
+                logger.error(f'Failed to expand url {url}')
+        # https://twitter.com/MeCookieMonster/status/1617921633456640001?s=20&t=3d0g4ZQis7dCbSDg-mE7-w
+        return self.link_clean_pattern.sub("\\1", url)
+
+    def is_rearchivable(self, url: str) -> bool:
+        # Twitter posts are static (for now)
+        return False
+
+    def download(self, item: Metadata) -> Metadata:
+        """
+        if this url is archivable will download post info and look for other posts from the same group with media.
+        can handle private/public channels
+        """
+        url = item.get_url()
+        # detect URLs that we definitely cannot handle
+        username, tweet_id = self.get_username_tweet_id(url)
+        if not username: return False
+
+        result = Metadata()
+
+        scr = TwitterTweetScraper(tweet_id)
+        try:
+            tweet = next(scr.get_items())
+        except Exception as ex:
+            logger.warning(f"can't get tweet: {type(ex).__name__} occurred. args: {ex.args}")
+            return self.download_alternative(item, url, tweet_id)
+
+        result.set_title(tweet.content).set_content(tweet.json()).set_timestamp(tweet.date)
+        if tweet.media is None:
+            logger.debug(f'No media found, archiving tweet text only')
+            return result
+
+        for i, tweet_media in enumerate(tweet.media):
+            media = Media(filename="")
+            mimetype = ""
+            if type(tweet_media) == Video:
+                variant = max(
+                    [v for v in tweet_media.variants if v.bitrate], key=lambda v: v.bitrate)
+                media.set("src", variant.url).set("duration", tweet_media.duration)
+                mimetype = variant.contentType
+            elif type(tweet_media) == Gif:
+                variant = tweet_media.variants[0]
+                media.set("src", variant.url)
+                mimetype = variant.contentType
+            elif type(tweet_media) == Photo:
+                media.set("src", tweet_media.fullUrl.replace('name=large', 'name=orig'))
+                mimetype = "image/jpeg"
+            else:
+                logger.warning(f"Could not get media URL of {tweet_media}")
+                continue
+            ext = mimetypes.guess_extension(mimetype)
+            media.filename = self.download_from_url(media.get("src"), f'{slugify(url)}_{i}{ext}', item)
+            result.add_media(media)
+
+        return result.success("twitter-snscrape")
+
+    def download_alternative(self, item: Metadata, url: str, tweet_id: str) -> Metadata:
+        """
+        CURRENTLY STOPPED WORKING
+        """
+        return False
+        # https://stackoverflow.com/a/71867055/6196010
+        logger.debug(f"Trying twitter hack for {url=}")
+        result = Metadata()
+
+        hack_url = f"https://cdn.syndication.twimg.com/tweet?id={tweet_id}"
+        r = requests.get(hack_url)
+        if r.status_code != 200: return False
+        tweet = r.json()
+
+        urls = []
+        for p in tweet["photos"]:
+            urls.append(p["url"])
+
+        # 1 tweet has 1 video max
+        if "video" in tweet:
+            v = tweet["video"]
+            urls.append(self.choose_variant(v.get("variants", [])))
+
+        logger.debug(f"Twitter hack got {urls=}")
+
+        for u in urls:
+            media = Media()
+            media.set("src", u)
+            media.filename = self.download_from_url(u, f'{slugify(url)}_{i}', item)
+            result.add_media(media)
+
+        result.set_content(json.dumps(tweet, ensure_ascii=False)).set_timestamp(datetime.strptime(tweet["created_at"], "%Y-%m-%dT%H:%M:%S.%fZ"))
+        return result
+
+    def get_username_tweet_id(self, url):
+        # detect URLs that we definitely cannot handle
+        matches = self.link_pattern.findall(url)
+        if not len(matches): return False, False
+
+        username, tweet_id = matches[0]  # only one URL supported
+        logger.debug(f"Found {username=} and {tweet_id=} in {url=}")
+
+        return username, tweet_id
+
+    def choose_variant(self, variants):
+        # choosing the highest quality possible
+        variant, width, height = None, 0, 0
+        for var in variants:
+            if var.get("type", "") == "video/mp4":
+                width_height = re.search(r"\/(\d+)x(\d+)\/", var["src"])
+                if width_height:
+                    w, h = int(width_height[1]), int(width_height[2])
+                    if w > width or h > height:
+                        width, height = w, h
+                        variant = var.get("src", variant)
+            else:
+                variant = var.get("src") if not variant else variant
+        return variant
--- a/src/auto_archiver/archivers/vk_archiver.py
+++ b/src/auto_archiver/archivers/vk_archiver.py
@@ -0,0 +1,57 @@
+from loguru import logger
+from vk_url_scraper import VkScraper
+
+from ..utils.misc import dump_payload
+from . import Archiver
+from ..core import Metadata, Media, ArchivingContext
+
+
+class VkArchiver(Archiver):
+    """"
+    VK videos are handled by YTDownloader, this archiver gets posts text and images.
+    Currently only works for /wall posts
+    """
+    name = "vk_archiver"
+
+    def __init__(self, config: dict) -> None:
+        super().__init__(config)
+        self.assert_valid_string("username")
+        self.assert_valid_string("password")
+        self.vks = VkScraper(self.username, self.password, session_file=self.session_file)
+
+    @staticmethod
+    def configs() -> dict:
+        return {
+            "username": {"default": None, "help": "valid VKontakte username"},
+            "password": {"default": None, "help": "valid VKontakte password"},
+            "session_file": {"default": "secrets/vk_config.v2.json", "help": "valid VKontakte password"},
+        }
+
+    def is_rearchivable(self, url: str) -> bool:
+        # VK content is static
+        return False
+
+    def download(self, item: Metadata) -> Metadata:
+        url = item.get_url()
+
+        if "vk.com" not in item.netloc: return False
+
+        # some urls can contain multiple wall/photo/... parts and all will be fetched
+        vk_scrapes = self.vks.scrape(url)
+        if not len(vk_scrapes): return False
+        logger.debug(f"VK: got {len(vk_scrapes)} scraped instances")
+
+        result = Metadata()
+        for scrape in vk_scrapes:
+            if not result.get_title():
+                result.set_title(scrape["text"])
+            if not result.get_timestamp():
+                result.set_timestamp(scrape["datetime"])
+
+        result.set_content(dump_payload(vk_scrapes))
+
+        filenames = self.vks.download_media(vk_scrapes, ArchivingContext.get_tmp_dir())
+        for filename in filenames:
+            result.add_media(Media(filename))
+
+        return result.success("vk")
--- a/src/auto_archiver/archivers/youtubedl_archiver.py
+++ b/src/auto_archiver/archivers/youtubedl_archiver.py
@@ -0,0 +1,67 @@
+import datetime, os, yt_dlp
+from loguru import logger
+
+from . import Archiver
+from ..core import Metadata, Media, ArchivingContext
+
+
+class YoutubeDLArchiver(Archiver):
+    name = "youtubedl_archiver"
+
+    def __init__(self, config: dict) -> None:
+        super().__init__(config)
+
+    @staticmethod
+    def configs() -> dict:
+        return {
+            "facebook_cookie": {"default": None, "help": "optional facebook cookie to have more access to content, from browser, looks like 'cookie: datr= xxxx'"},
+        }
+
+    def download(self, item: Metadata) -> Metadata:
+        #TODO: yt-dlp for transcripts?
+        url = item.get_url()
+
+        if item.netloc in ['facebook.com', 'www.facebook.com'] and self.facebook_cookie:
+            logger.debug('Using Facebook cookie')
+            yt_dlp.utils.std_headers['cookie'] = self.facebook_cookie
+
+        ydl = yt_dlp.YoutubeDL({'outtmpl': os.path.join(ArchivingContext.get_tmp_dir(), f'%(id)s.%(ext)s'), 'quiet': False})
+
+        try:
+            # don'd download since it can be a live stream
+            info = ydl.extract_info(url, download=False)
+            if info.get('is_live', False):
+                logger.warning("Live streaming media, not archiving now")
+                return False
+        except yt_dlp.utils.DownloadError as e:
+            logger.debug(f'No video - Youtube normal control flow: {e}')
+            return False
+        except Exception as e:
+            logger.debug(f'ytdlp exception which is normal for example a facebook page with images only will cause a IndexError: list index out of range. Exception here is: \n  {e}')
+            return False
+
+        # this time download
+        info = ydl.extract_info(url, download=True)
+        if "entries" in info:
+            entries = info.get("entries", [])
+            if not len(entries):
+                logger.warning('YoutubeDLArchiver could not find any video')
+                return False
+        else: entries = [info]
+
+        result = Metadata()
+        result.set_title(info.get("title"))
+        for entry in entries:
+            filename = ydl.prepare_filename(entry)
+            if not os.path.exists(filename):
+                filename = filename.split('.')[0] + '.mkv'
+            result.add_media(Media(filename).set("duration", info.get("duration")))
+
+        if (timestamp := info.get("timestamp")):
+            timestamp = datetime.datetime.utcfromtimestamp(timestamp).replace(tzinfo=datetime.timezone.utc).isoformat()
+            result.set_timestamp(timestamp)
+        if (upload_date := info.get("upload_date")):
+            upload_date = datetime.datetime.strptime(upload_date, '%Y%m%d').replace(tzinfo=datetime.timezone.utc)
+            result.set("upload_date", upload_date)
+
+        return result.success("yt-dlp")
--- a/src/auto_archiver/core/init.py
+++ b/src/auto_archiver/core/init.py
@@ -0,0 +1,8 @@
+from .metadata import Metadata
+from .media import Media
+from .step import Step
+from .context import ArchivingContext
+
+# cannot import ArchivingOrchestrator/Config to avoid circular dep
+# from .orchestrator import ArchivingOrchestrator
+# from .config import Config
--- a/src/auto_archiver/core/config.py
+++ b/src/auto_archiver/core/config.py
@@ -0,0 +1,117 @@
+
+
+import argparse, yaml
+from dataclasses import dataclass, field
+from typing import List
+from collections import defaultdict
+from loguru import logger
+
+from ..archivers import Archiver
+from ..feeders import Feeder
+from ..databases import Database
+from ..formatters import Formatter
+from ..storages import Storage
+from ..enrichers import Enricher
+from . import Step
+
+
+@dataclass
+class Config:
+    configurable_parents = [
+        Feeder,
+        Enricher,
+        Archiver,
+        Database,
+        Storage,
+        Formatter
+        # Util
+    ]
+    feeder: Feeder
+    formatter: Formatter
+    archivers: List[Archiver] = field(default_factory=[])
+    enrichers: List[Enricher] = field(default_factory=[])
+    storages: List[Storage] = field(default_factory=[])
+    databases: List[Database] = field(default_factory=[])
+
+    def __init__(self) -> None:
+        self.defaults = {}
+        self.cli_ops = {}
+        self.config = {}
+
+    def parse(self, use_cli=True, yaml_config_filename: str = None):
+        """
+        if yaml_config_filename is provided, the --config argument is ignored, 
+        useful for library usage when the config values are preloaded
+        """
+        # 1. parse CLI values
+        if use_cli:
+            parser = argparse.ArgumentParser(
+                # prog = "auto-archiver",
+                description="Auto Archiver is a CLI tool to archive media/metadata from online URLs; it can read URLs from many sources (Google Sheets, Command Line, ...); and write results to many destinations too (CSV, Google Sheets, MongoDB, ...)!",
+                epilog="Check the code at https://github.com/bellingcat/auto-archiver"
+            )
+
+            parser.add_argument('--config', action='store', dest='config', help='the filename of the YAML configuration file (defaults to \'config.yaml\')', default='orchestration.yaml')
+
+        for configurable in self.configurable_parents:
+            child: Step
+            for child in configurable.__subclasses__():
+                assert child.configs() is not None and type(child.configs()) == dict, f"class '{child.name}' should have a configs method returning a dict."
+                for config, details in child.configs().items():
+                    assert "." not in child.name, f"class prop name cannot contain dots('.'): {child.name}"
+                    assert "." not in config, f"config property cannot contain dots('.'): {config}"
+                    config_path = f"{child.name}.{config}"
+
+                    if use_cli:
+                        try:
+                            parser.add_argument(f'--{config_path}', action='store', dest=config_path, help=f"{details['help']} (defaults to {details['default']})", choices=details.get("choices", None))
+                        except argparse.ArgumentError:
+                            # captures cases when a Step is used in 2 flows, eg: wayback enricher vs wayback archiver
+                            pass
+
+                    self.defaults[config_path] = details["default"]
+                    if "cli_set" in details:
+                        self.cli_ops[config_path] = details["cli_set"]
+
+        if use_cli:
+            args = parser.parse_args()
+            yaml_config_filename = yaml_config_filename or getattr(args, "config")
+        else: args = {}
+
+        # 2. read YAML config file (or use provided value)
+        self.yaml_config = self.read_yaml(yaml_config_filename)
+
+        # 3. CONFIGS: decide value with priority: CLI >> config.yaml >> default
+        self.config = defaultdict(dict)
+        for config_path, default in self.defaults.items():
+            child, config = tuple(config_path.split("."))
+            val = getattr(args, config_path, None)
+            if val is not None and config_path in self.cli_ops:
+                val = self.cli_ops[config_path](val, default)
+            if val is None:
+                val = self.yaml_config.get("configurations", {}).get(child, {}).get(config, default)
+            self.config[child][config] = val
+        self.config = dict(self.config)
+
+        # 4. STEPS: read steps and validate they exist
+        steps = self.yaml_config.get("steps", {})
+        assert "archivers" in steps, "your configuration steps are missing the archivers property"
+        assert "storages" in steps, "your configuration steps are missing the storages property"
+
+        self.feeder = Feeder.init(steps.get("feeder", "cli_feeder"), self.config)
+        self.formatter = Formatter.init(steps.get("formatter", "mute_formatter"), self.config)
+        self.enrichers = [Enricher.init(e, self.config) for e in steps.get("enrichers", [])]
+        self.archivers = [Archiver.init(e, self.config) for e in (steps.get("archivers") or [])]
+        self.databases = [Database.init(e, self.config) for e in steps.get("databases", [])]
+        self.storages = [Storage.init(e, self.config) for e in steps.get("storages", [])]
+
+        logger.info(f"FEEDER: {self.feeder.name}")
+        logger.info(f"ENRICHERS: {[x.name for x in self.enrichers]}")
+        logger.info(f"ARCHIVERS: {[x.name for x in self.archivers]}")
+        logger.info(f"DATABASES: {[x.name for x in self.databases]}")
+        logger.info(f"STORAGES: {[x.name for x in self.storages]}")
+        logger.info(f"FORMATTER: {self.formatter.name}")
+
+    def read_yaml(self, yaml_filename: str) -> dict:
+        with open(yaml_filename, "r", encoding="utf-8") as inf:
+            return yaml.safe_load(inf)
--- a/src/auto_archiver/core/context.py
+++ b/src/auto_archiver/core/context.py
@@ -0,0 +1,52 @@
+from loguru import logger
+
+
+class ArchivingContext:
+    """
+    Singleton context class.
+    ArchivingContext._get_instance() to retrieve it if needed
+    otherwise just 
+    ArchivingContext.set(key, value)
+    and 
+    ArchivingContext.get(key, default)
+
+    When reset is called, all values are cleared EXCEPT if they were .set(keep_on_reset=True)
+        reset(full_reset=True) will recreate everything including the keep_on_reset status
+    """
+    _instance = None
+
+    def __init__(self):
+        self.configs = {}
+        self.keep_on_reset = set()
+
+    @staticmethod
+    def get_instance():
+        if ArchivingContext._instance is None:
+            ArchivingContext._instance = ArchivingContext()
+        return ArchivingContext._instance
+
+    @staticmethod
+    def set(key, value, keep_on_reset: bool = False):
+        ac = ArchivingContext.get_instance()
+        ac.configs[key] = value
+        if keep_on_reset: ac.keep_on_reset.add(key)
+
+    @staticmethod
+    def get(key: str, default=None):
+        return ArchivingContext.get_instance().configs.get(key, default)
+
+    @staticmethod
+    def reset(full_reset: bool = False):
+        ac = ArchivingContext.get_instance()
+        if full_reset: ac.keep_on_reset = set()
+        ac.configs = {k: v for k, v in ac.configs.items() if k in ac.keep_on_reset}
+
+    # ---- custom getters/setters for widely used context values
+
+    @staticmethod
+    def set_tmp_dir(tmp_dir: str):
+        ArchivingContext.get_instance().configs["tmp_dir"] = tmp_dir
+
+    @staticmethod
+    def get_tmp_dir() -> str:
+        return ArchivingContext.get_instance().configs.get("tmp_dir")
--- a/src/auto_archiver/core/media.py
+++ b/src/auto_archiver/core/media.py
@@ -0,0 +1,73 @@
+
+from __future__ import annotations
+from ast import List
+from typing import Any
+from dataclasses import dataclass, field
+from dataclasses_json import dataclass_json, config
+import mimetypes
+
+from .context import ArchivingContext
+
+from loguru import logger
+
+
+@dataclass_json  # annotation order matters
+@dataclass
+class Media:
+    filename: str
+    key: str = None
+    urls: List[str] = field(default_factory=list)
+    properties: dict = field(default_factory=dict)
+    _mimetype: str = None  # eg: image/jpeg
+    _stored: bool = field(default=False, repr=False, metadata=config(exclude=lambda _: True))  # always exclude
+
+    def store(self: Media, override_storages: List = None, url: str = "url-not-available"):
+        # stores the media into the provided/available storages [Storage]
+        # repeats the process for its properties, in case they have inner media themselves
+        # for now it only goes down 1 level but it's easy to make it recursive if needed
+        storages = override_storages or ArchivingContext.get("storages")
+        if not len(storages):
+            logger.warning(f"No storages found in local context or provided directly for {self.filename}.")
+            return
+
+        for s in storages:
+            s.store(self, url)
+            # Media can be inside media properties, examples include transformations on original media
+            for prop in self.properties.values():
+                if isinstance(prop, Media):
+                    s.store(prop, url)
+                if isinstance(prop, list):
+                    for prop_media in prop:
+                        if isinstance(prop_media, Media):
+                            s.store(prop_media, url)
+
+    def is_stored(self) -> bool:
+        return len(self.urls) > 0 and len(self.urls) == len(ArchivingContext.get("storages"))
+
+    def set(self, key: str, value: Any) -> Media:
+        self.properties[key] = value
+        return self
+
+    def get(self, key: str, default: Any = None) -> Any:
+        return self.properties.get(key, default)
+
+    def add_url(self, url: str) -> None:
+        # url can be remote, local, ...
+        self.urls.append(url)
+
+    @property  # getter .mimetype
+    def mimetype(self) -> str:
+        assert self.filename is not None and len(self.filename) > 0, "cannot get mimetype from media without filename"
+        if not self._mimetype:
+            self._mimetype = mimetypes.guess_type(self.filename)[0]
+        return self._mimetype or ""
+
+    @mimetype.setter  # setter .mimetype
+    def mimetype(self, v: str) -> None:
+        self._mimetype = v
+
+    def is_video(self) -> bool:
+        return self.mimetype.startswith("video")
+
+    def is_audio(self) -> bool:
+        return self.mimetype.startswith("audio")
--- a/src/auto_archiver/core/metadata.py
+++ b/src/auto_archiver/core/metadata.py
@@ -0,0 +1,142 @@
+
+from __future__ import annotations
+from ast import List, Set
+from typing import Any, Union, Dict
+from dataclasses import dataclass, field
+from dataclasses_json import dataclass_json, config
+import datetime
+from urllib.parse import urlparse
+from dateutil.parser import parse as parse_dt
+from .media import Media
+from .context import ArchivingContext
+
+
+@dataclass_json  # annotation order matters
+@dataclass
+class Metadata:
+    status: str = "no archiver"
+    metadata: Dict[str, Any] = field(default_factory=dict)
+    media: List[Media] = field(default_factory=list)
+    rearchivable: bool = True  # defaults to true, archivers can overwrite
+
+    def __post_init__(self):
+        self.set("_processed_at", datetime.datetime.utcnow())
+
+    def merge(self: Metadata, right: Metadata, overwrite_left=True) -> Metadata:
+        """
+        merges two Metadata instances, will overwrite according to overwrite_left flag
+        """
+        if not right: return self
+        if overwrite_left:
+            if right.status and len(right.status):
+                self.status = right.status
+            self.rearchivable |= right.rearchivable
+            for k, v in right.metadata.items():
+                assert k not in self.metadata or type(v) == type(self.get(k))
+                if type(v) not in [dict, list, set] or k not in self.metadata:
+                    self.set(k, v)
+                else:  # key conflict
+                    if type(v) in [dict, set]: self.set(k, self.get(k) | v)
+                    elif type(v) == list: self.set(k, self.get(k) + v)
+            self.media.extend(right.media)
+        else:  # invert and do same logic
+            return right.merge(self)
+        return self
+
+    def store(self: Metadata, override_storages: List = None):
+        # calls .store for all contained media. storages [Storage]
+        storages = override_storages or ArchivingContext.get("storages")
+        for media in self.media:
+            media.store(override_storages=storages, url=self.get_url())
+
+    def set(self, key: str, val: Any) -> Metadata:
+        self.metadata[key] = val
+        return self
+
+    def get(self, key: str, default: Any = None, create_if_missing=False) -> Union[Metadata, str]:
+        # goes through metadata and returns the Metadata available
+        if create_if_missing and key not in self.metadata:
+            self.metadata[key] = default
+        return self.metadata.get(key, default)
+
+    def success(self, context: str = None) -> Metadata:
+        if context: self.status = f"{context}: success"
+        else: self.status = "success"
+        return self
+
+    def is_success(self) -> bool:
+        return "success" in self.status
+
+    def is_empty(self) -> bool:
+        return not self.is_success() and len(self.media) == 0 and len(self.metadata) <= 2  # url, processed_at
+
+    @property  # getter .netloc
+    def netloc(self) -> str:
+        return urlparse(self.get_url()).netloc
+
+
+# custom getter/setters
+
+
+    def set_url(self, url: str) -> Metadata:
+        assert type(url) is str and len(url) > 0, "invalid URL"
+        return self.set("url", url)
+
+    def get_url(self) -> str:
+        url = self.get("url")
+        assert type(url) is str and len(url) > 0, "invalid URL"
+        return url
+
+    def set_content(self, content: str) -> Metadata:
+        # a dump with all the relevant content
+        append_content = (self.get("content", "") + content + "\n").strip()
+        return self.set("content", append_content)
+
+    def set_title(self, title: str) -> Metadata:
+        return self.set("title", title)
+
+    def get_title(self) -> str:
+        return self.get("title")
+
+    def set_timestamp(self, timestamp: datetime.datetime) -> Metadata:
+        if type(timestamp) == str:
+            timestamp = parse_dt(timestamp)
+        assert type(timestamp) == datetime.datetime, "set_timestamp expects a datetime instance"
+        return self.set("timestamp", timestamp)
+
+    def get_timestamp(self, utc=True, iso=True) -> datetime.datetime:
+        ts = self.get("timestamp")
+        if not ts: return ts
+        if utc: ts = ts.replace(tzinfo=datetime.timezone.utc)
+        if iso: return ts.isoformat()
+        return ts
+
+    def add_media(self, media: Media, id: str = None) -> Metadata:
+        # adds a new media, optionally including an id
+        if media is None: return
+        if id is not None:
+            assert not len([1 for m in self.media if m.get("id") == id]), f"cannot add 2 pieces of media with the same id {id}"
+            media.set("id", id)
+        self.media.append(media)
+        return media
+
+    def get_media_by_id(self, id: str, default=None) -> Media:
+        for m in self.media:
+            if m.get("id") == id: return m
+        return default
+
+    def get_first_image(self, default=None) -> Media:
+        for m in self.media:
+            if "image" in m.mimetype: return m
+        return default
+
+    def set_final_media(self, final: Media) -> Metadata:
+        """final media is a special type of media: if you can show only 1 this is it, it's useful for some DBs like GsheetDb"""
+        self.add_media(final, "_final_media")
+
+    def get_final_media(self) -> Media:
+        _default = self.media[0] if len(self.media) else None
+        return self.get_media_by_id("_final_media", _default)
+
+    def __str__(self) -> str:
+        return self.__repr__()
--- a/src/auto_archiver/core/orchestrator.py
+++ b/src/auto_archiver/core/orchestrator.py
@@ -0,0 +1,126 @@
+from __future__ import annotations
+from ast import List
+from typing import Union
+
+from .context import ArchivingContext
+
+from ..archivers import Archiver
+from ..feeders import Feeder
+from ..formatters import Formatter
+from ..storages import Storage
+from ..enrichers import Enricher
+from ..databases import Database
+from .media import Media
+from .metadata import Metadata
+
+import tempfile, traceback
+from loguru import logger
+
+
+class ArchivingOrchestrator:
+    def __init__(self, config) -> None:
+        self.feeder: Feeder = config.feeder
+        self.formatter: Formatter = config.formatter
+        self.enrichers: List[Enricher] = config.enrichers
+        self.archivers: List[Archiver] = config.archivers
+        self.databases: List[Database] = config.databases
+        self.storages: List[Storage] = config.storages
+        ArchivingContext.set("storages", self.storages, keep_on_reset=True)
+
+        for a in self.archivers: a.setup()
+
+    def feed(self) -> None:
+        for item in self.feeder:
+            self.feed_item(item)
+
+    def feed_item(self, item: Metadata) -> Metadata:
+        try:
+            ArchivingContext.reset()
+            with tempfile.TemporaryDirectory(dir="./") as tmp_dir:
+                ArchivingContext.set_tmp_dir(tmp_dir)
+                return self.archive(item)
+        except KeyboardInterrupt:
+            # catches keyboard interruptions to do a clean exit
+            logger.warning(f"caught interrupt on {item=}")
+            for d in self.databases: d.aborted(item)
+            exit()
+        except Exception as e:
+            logger.error(f'Got unexpected error on item {item}: {e}\n{traceback.format_exc()}')
+            for d in self.databases: d.failed(item)
+
+        # how does this handle the parameters like folder which can be different for each archiver?
+        # the storage needs to know where to archive!!
+        # solution: feeders have context: extra metadata that they can read or ignore,
+        # all of it should have sensible defaults (eg: folder)
+        # default feeder is a list with 1 element
+
+    def archive(self, result: Metadata) -> Union[Metadata, None]:
+        original_url = result.get_url()
+
+        # 1 - cleanup
+        # each archiver is responsible for cleaning/expanding its own URLs
+        url = original_url
+        for a in self.archivers: url = a.sanitize_url(url)
+        result.set_url(url)
+        if original_url != url: result.set("original_url", original_url)
+
+        # 2 - rearchiving logic + notify start to DB
+        # archivers can signal whether the content is rearchivable: eg: tweet vs webpage
+        for a in self.archivers: result.rearchivable |= a.is_rearchivable(url)
+        logger.debug(f"{result.rearchivable=} for {url=}")
+
+        # signal to DB that archiving has started
+        # and propagate already archived if it exists
+        cached_result = None
+        for d in self.databases:
+            # are the databases to decide whether to archive?
+            # they can simply return True by default, otherwise they can avoid duplicates. should this logic be more granular, for example on the archiver level: a tweet will not need be scraped twice, whereas an instagram profile might. the archiver could not decide from the link which parts to archive,
+            # instagram profile example: it would always re-archive everything
+            # maybe the database/storage could use a hash/key to decide if there's a need to re-archive
+            d.started(result)
+            if (local_result := d.fetch(result)):
+                cached_result = (cached_result or Metadata()).merge(local_result)
+        if cached_result and not cached_result.rearchivable:
+            logger.debug("Found previously archived entry")
+            for d in self.databases:
+                d.done(cached_result)
+            return cached_result
+
+        # 3 - call archivers until one succeeds
+        for a in self.archivers:
+            logger.info(f"Trying archiver {a.name} for {url}")
+            try:
+                # Q: should this be refactored so it's just a.download(result)?
+                result.merge(a.download(result))
+                if result.is_success(): break
+            except Exception as e: logger.error(f"Unexpected error with archiver {a.name}: {e}: {traceback.format_exc()}")
+
+        # what if an archiver returns multiple entries and one is to be part of HTMLgenerator?
+        # should it call the HTMLgenerator as if it's not an enrichment?
+        # eg: if it is enable: generates an HTML with all the returned media, should it include enrichers? yes
+        # then how to execute it last? should there also be post-processors? are there other examples?
+        # maybe as a PDF? or a Markdown file
+
+        # 4 - call enrichers: have access to archived content, can generate metadata and Media
+        # eg: screenshot, wacz, webarchive, thumbnails
+        for e in self.enrichers:
+            try: e.enrich(result)
+            except Exception as exc: logger.error(f"Unexpected error with enricher {e.name}: {exc}: {traceback.format_exc()}")
+
+        # 5 - store media
+        # looks for Media in result.media and also result.media[x].properties (as list or dict values)
+        result.store()
+
+        # 6 - format and store formatted if needed
+        # enrichers typically need access to already stored URLs etc
+        if (final_media := self.formatter.format(result)):
+            final_media.store(url=url)
+            result.set_final_media(final_media)
+
+        if result.is_empty():
+            result.status = "nothing archived"
+
+        # signal completion to databases (DBs, Google Sheets, CSV, ...)
+        for d in self.databases: d.done(result)
+
+        return result
--- a/src/auto_archiver/core/step.py
+++ b/src/auto_archiver/core/step.py
@@ -0,0 +1,38 @@
+from __future__ import annotations
+from dataclasses import dataclass, field
+from inspect import ClassFoundException
+from typing import Type
+from abc import ABC
+# from collections.abc import Iterable
+
+
+@dataclass
+class Step(ABC):
+    name: str = None
+
+    def __init__(self, config: dict) -> None:
+        # reads the configs into object properties
+        # self.config = config[self.name]
+        for k, v in config.get(self.name, {}).items():
+            self.__setattr__(k, v)
+
+    @staticmethod
+    def configs() -> dict: return {}
+
+    def init(name: str, config: dict, child: Type[Step]) -> Step:
+        """
+        looks into direct subclasses of child for name and returns such ab object
+        TODO: cannot find subclasses of child.subclasses
+        """
+        for sub in child.__subclasses__():
+            if sub.name == name:
+                return sub(config)
+        raise ClassFoundException(f"Unable to initialize STEP with {name=}, check your configuration file/step names, and make sure you made the step discoverable by putting it into __init__.py")
+
+    def assert_valid_string(self, prop: str) -> None:
+        """
+        receives a property name an ensures it exists and is a valid non-empty string, raises an exception if not
+        """
+        assert hasattr(self, prop), f"property {prop} not found"
+        s = getattr(self, prop)
+        assert s is not None and type(s) == str and len(s) > 0, f"invalid property {prop} value '{s}', it should be a valid string"
--- a/src/auto_archiver/databases/init.py
+++ b/src/auto_archiver/databases/init.py
@@ -0,0 +1,4 @@
+from .database import Database
+from .gsheet_db import GsheetsDb
+from .console_db import ConsoleDb
+from .csv_db import CSVDb
--- a/src/auto_archiver/databases/console_db.py
+++ b/src/auto_archiver/databases/console_db.py
@@ -0,0 +1,32 @@
+from loguru import logger
+
+from . import Database
+from ..core import Metadata
+
+
+class ConsoleDb(Database):
+    """
+        Outputs results to the console
+    """
+    name = "console_db"
+
+    def __init__(self, config: dict) -> None:
+        # without this STEP.__init__ is not called
+        super().__init__(config)
+
+    @staticmethod
+    def configs() -> dict:
+        return {}
+
+    def started(self, item: Metadata) -> None:
+        logger.warning(f"STARTED {item}")
+
+    def failed(self, item: Metadata) -> None:
+        logger.error(f"FAILED {item}")
+
+    def aborted(self, item: Metadata) -> None:
+        logger.warning(f"ABORTED {item}")
+
+    def done(self, item: Metadata) -> None:
+        """archival result ready - should be saved to DB"""
+        logger.success(f"DONE {item}")
--- a/src/auto_archiver/databases/csv_db.py
+++ b/src/auto_archiver/databases/csv_db.py
@@ -0,0 +1,34 @@
+import os
+from loguru import logger
+from csv import DictWriter
+from dataclasses import asdict
+
+from . import Database
+from ..core import Metadata
+
+
+class CSVDb(Database):
+    """
+        Outputs results to a CSV file
+    """
+    name = "csv_db"
+
+    def __init__(self, config: dict) -> None:
+        # without this STEP.__init__ is not called
+        super().__init__(config)
+        self.assert_valid_string("csv_file")
+
+    @staticmethod
+    def configs() -> dict:
+        return {
+            "csv_file": {"default": "db.csv", "help": "CSV file name"}
+        }
+
+    def done(self, item: Metadata) -> None:
+        """archival result ready - should be saved to DB"""
+        logger.success(f"DONE {item}")
+        is_empty = not os.path.isfile(self.csv_file) or os.path.getsize(self.csv_file) == 0
+        with open(self.csv_file, "a", encoding="utf-8") as outf:
+            writer = DictWriter(outf, fieldnames=asdict(Metadata()))
+            if is_empty: writer.writeheader()
+            writer.writerow(asdict(item))
--- a/src/auto_archiver/databases/database.py
+++ b/src/auto_archiver/databases/database.py
@@ -0,0 +1,41 @@
+from __future__ import annotations
+from dataclasses import dataclass
+from abc import abstractmethod, ABC
+from typing import Union
+
+from ..core import Metadata, Step
+
+
+@dataclass
+class Database(Step, ABC):
+    name = "database"
+
+    def __init__(self, config: dict) -> None:
+        # without this STEP.__init__ is not called
+        super().__init__(config)
+
+    def init(name: str, config: dict) -> Database:
+        # only for typing...
+        return Step.init(name, config, Database)
+
+    def started(self, item: Metadata) -> None:
+        """signals the DB that the given item archival has started"""
+        pass
+
+    def failed(self, item: Metadata) -> None:
+        """update DB accordingly for failure"""
+        pass
+
+    def aborted(self, item: Metadata) -> None:
+        """abort notification if user cancelled after start"""
+        pass
+
+    # @abstractmethod
+    def fetch(self, item: Metadata) -> Union[Metadata, bool]:
+        """check if the given item has been archived already"""
+        return False
+
+    @abstractmethod
+    def done(self, item: Metadata) -> None:
+        """archival result ready - should be saved to DB"""
+        pass
--- a/src/auto_archiver/databases/gsheet_db.py
+++ b/src/auto_archiver/databases/gsheet_db.py
@@ -0,0 +1,92 @@
+from typing import Union, Tuple
+import datetime
+from urllib.parse import quote
+
+from loguru import logger
+
+from . import Database
+from ..core import Metadata, Media, ArchivingContext
+from ..utils import GWorksheet
+
+
+class GsheetsDb(Database):
+    """
+        NB: only works if GsheetFeeder is used. 
+        could be updated in the future to support non-GsheetFeeder metadata 
+    """
+    name = "gsheet_db"
+
+    def __init__(self, config: dict) -> None:
+        # without this STEP.__init__ is not called
+        super().__init__(config)
+
+    @staticmethod
+    def configs() -> dict:
+        return {}
+
+    def started(self, item: Metadata) -> None:
+        logger.warning(f"STARTED {item}")
+        gw, row = self._retrieve_gsheet(item)
+        gw.set_cell(row, 'status', 'Archive in progress')
+
+    def failed(self, item: Metadata) -> None:
+        logger.error(f"FAILED {item}")
+        self._safe_status_update(item, 'Archive failed')
+
+    def aborted(self, item: Metadata) -> None:
+        logger.warning(f"ABORTED {item}")
+        self._safe_status_update(item, '')
+
+    def fetch(self, item: Metadata) -> Union[Metadata, bool]:
+        """check if the given item has been archived already"""
+        return False
+
+    def done(self, item: Metadata) -> None:
+        """archival result ready - should be saved to DB"""
+        logger.success(f"DONE {item.get_url()}")
+        gw, row = self._retrieve_gsheet(item)
+        # self._safe_status_update(item, 'done')
+
+        cell_updates = []
+        row_values = gw.get_row(row)
+
+        def batch_if_valid(col, val, final_value=None):
+            final_value = final_value or val
+            if val and gw.col_exists(col) and gw.get_cell(row_values, col) == '':
+                cell_updates.append((row, col, final_value))
+
+        cell_updates.append((row, 'status', item.status))
+
+        media: Media = item.get_final_media()
+        if hasattr(media, "urls"):
+            batch_if_valid('archive', "\n".join(media.urls))
+        batch_if_valid('date', True, datetime.datetime.utcnow().replace(tzinfo=datetime.timezone.utc).isoformat())
+        batch_if_valid('title', item.get_title())
+        batch_if_valid('text', item.get("content", ""))
+        batch_if_valid('timestamp', item.get_timestamp())
+        batch_if_valid('hash', media.get("hash", "not-calculated"))
+        if (screenshot := item.get_media_by_id("screenshot")) and hasattr(screenshot, "urls"):
+            batch_if_valid('screenshot', "\n".join(screenshot.urls))
+
+        if (thumbnail := item.get_first_image("thumbnail")):
+            if hasattr(thumbnail, "urls"):
+                batch_if_valid('thumbnail', f'=IMAGE("{thumbnail.urls[0]}")')
+
+        if (browsertrix := item.get_media_by_id("browsertrix")):
+            batch_if_valid('wacz', "\n".join(browsertrix.urls))
+            batch_if_valid('replaywebpage', "\n".join([f'https://replayweb.page/?source={quote(wacz)}#view=pages&url={quote(item.get_url())}' for wacz in browsertrix.urls]))
+
+        gw.batch_set_cell(cell_updates)
+
+    def _safe_status_update(self, item: Metadata, new_status: str) -> None:
+        try:
+            gw, row = self._retrieve_gsheet(item)
+            gw.set_cell(row, 'status', new_status)
+        except Exception as e:
+            logger.debug(f"Unable to update sheet: {e}")
+
+    def _retrieve_gsheet(self, item: Metadata) -> Tuple[GWorksheet, int]:
+        # TODO: to make gsheet_db less coupled with gsheet_feeder's "gsheet" parameter, this method could 1st try to fetch "gsheet" from ArchivingContext and, if missing, manage its own singleton - not needed for now
+        gw: GWorksheet = ArchivingContext.get("gsheet").get("worksheet")
+        row: int = ArchivingContext.get("gsheet").get("row")
+        return gw, row
--- a/src/auto_archiver/enrichers/init.py
+++ b/src/auto_archiver/enrichers/init.py
@@ -0,0 +1,7 @@
+from .enricher import Enricher
+from .screenshot_enricher import ScreenshotEnricher 
+from .wayback_enricher import WaybackArchiverEnricher
+from .hash_enricher import HashEnricher
+from .thumbnail_enricher import ThumbnailEnricher
+from .wacz_enricher import WaczEnricher
+from .whisper_enricher import WhisperEnricher
--- a/src/auto_archiver/enrichers/enricher.py
+++ b/src/auto_archiver/enrichers/enricher.py
@@ -0,0 +1,20 @@
+from __future__ import annotations
+from dataclasses import dataclass
+from abc import abstractmethod, ABC
+from ..core import Metadata, Step
+
+@dataclass
+class Enricher(Step, ABC):
+    name = "enricher"
+
+    def __init__(self, config: dict) -> None:
+        # without this STEP.__init__ is not called
+        super().__init__(config)
+        
+
+    # only for typing...
+    def init(name: str, config: dict) -> Enricher:
+        return Step.init(name, config, Enricher)
+
+    @abstractmethod
+    def enrich(self, to_enrich: Metadata) -> None: pass
--- a/src/auto_archiver/enrichers/hash_enricher.py
+++ b/src/auto_archiver/enrichers/hash_enricher.py
@@ -0,0 +1,49 @@
+import hashlib
+from loguru import logger
+
+from . import Enricher
+from ..core import Metadata, ArchivingContext
+
+
+class HashEnricher(Enricher):
+    """
+    Calculates hashes for Media instances
+    """
+    name = "hash_enricher"
+
+    def __init__(self, config: dict) -> None:
+        # without this STEP.__init__ is not called
+        super().__init__(config)
+        algo_choices = self.configs()["algorithm"]["choices"]
+        assert self.algorithm in algo_choices, f"Invalid hash algorithm selected, must be one of {algo_choices} (you selected {self.algorithm})."
+        self.chunksize = int(self.chunksize)
+        ArchivingContext.set("hash_enricher.algorithm", self.algorithm, keep_on_reset=True)
+
+    @staticmethod
+    def configs() -> dict:
+        return {
+            "algorithm": {"default": "SHA-256", "help": "hash algorithm to use", "choices": ["SHA-256", "SHA3-512"]},
+            "chunksize": {"default": 1.6e7, "help": "number of bytes to use when reading files in chunks (if this value is too large you will run out of RAM), default is 16MB"},
+        }
+
+    def enrich(self, to_enrich: Metadata) -> None:
+        url = to_enrich.get_url()
+        logger.debug(f"calculating media hashes for {url=} (using {self.algorithm})")
+
+        for i, m in enumerate(to_enrich.media):
+            if len(hd := self.calculate_hash(m.filename)):
+                to_enrich.media[i].set("hash", f"{self.algorithm}:{hd}")
+
+    def calculate_hash(self, filename):
+        hash = None
+        if self.algorithm == "SHA-256":
+            hash = hashlib.sha256()
+        elif self.algorithm == "SHA3-512":
+            hash = hashlib.sha3_512()
+        else: return ""
+        with open(filename, "rb") as f:
+            while True:
+                buf = f.read(self.chunksize)
+                if not buf: break
+                hash.update(buf)
+        return hash.hexdigest()
--- a/src/auto_archiver/enrichers/screenshot_enricher.py
+++ b/src/auto_archiver/enrichers/screenshot_enricher.py
@@ -0,0 +1,38 @@
+from loguru import logger
+import time, uuid, os
+from selenium.common.exceptions import TimeoutException
+
+from . import Enricher
+from ..utils import Webdriver, UrlUtil
+from ..core import Media, Metadata, ArchivingContext
+
+class ScreenshotEnricher(Enricher):
+    name = "screenshot_enricher"
+
+    @staticmethod
+    def configs() -> dict:
+        return {
+            "width": {"default": 1280, "help": "width of the screenshots"},
+            "height": {"default": 720, "help": "height of the screenshots"},
+            "timeout": {"default": 60, "help": "timeout for taking the screenshot"},
+            "sleep_before_screenshot": {"default": 4, "help": "seconds to wait for the pages to load before taking screenshot"}
+        }
+
+    def enrich(self, to_enrich: Metadata) -> None:
+        url = to_enrich.get_url()
+        if UrlUtil.is_auth_wall(url):
+            logger.debug(f"[SKIP] SCREENSHOT since url is behind AUTH WALL: {url=}")
+            return
+
+        logger.debug(f"Enriching screenshot for {url=}")
+        with Webdriver(self.width, self.height, self.timeout, 'facebook.com' in url) as driver:
+            try:
+                driver.get(url)
+                time.sleep(int(self.sleep_before_screenshot))
+                screenshot_file = os.path.join(ArchivingContext.get_tmp_dir(), f"screenshot_{str(uuid.uuid4())[0:8]}.png")
+                driver.save_screenshot(screenshot_file)
+                to_enrich.add_media(Media(filename=screenshot_file), id="screenshot")
+            except TimeoutException:
+                logger.info("TimeoutException loading page for screenshot")
+            except Exception as e:
+                logger.error(f"Got error while loading webdriver for screenshot enricher: {e}")
--- a/src/auto_archiver/enrichers/thumbnail_enricher.py
+++ b/src/auto_archiver/enrichers/thumbnail_enricher.py
@@ -0,0 +1,45 @@
+import ffmpeg, os, uuid
+from loguru import logger
+
+from . import Enricher
+from ..core import Media, Metadata, ArchivingContext
+
+
+class ThumbnailEnricher(Enricher):
+    """
+    Generates thumbnails for all the media
+    """
+    name = "thumbnail_enricher"
+
+    def __init__(self, config: dict) -> None:
+        # without this STEP.__init__ is not called
+        super().__init__(config)
+
+    @staticmethod
+    def configs() -> dict:
+        return {}
+
+    def enrich(self, to_enrich: Metadata) -> None:
+        logger.debug(f"generating thumbnails")
+        for i, m in enumerate(to_enrich.media[::]):
+            if m.is_video():
+                folder = os.path.join(ArchivingContext.get_tmp_dir(), str(uuid.uuid4()))
+                os.makedirs(folder, exist_ok=True)
+                logger.debug(f"generating thumbnails for {m.filename}")
+                fps, duration = 0.5, m.get("duration")
+                if duration is not None:
+                    duration = float(duration)
+                    if duration < 60: fps = 10.0 / duration
+                    elif duration < 120: fps = 20.0 / duration
+                    else: fps = 40.0 / duration
+
+                stream = ffmpeg.input(m.filename)
+                stream = ffmpeg.filter(stream, 'fps', fps=fps).filter('scale', 512, -1)
+                stream.output(os.path.join(folder, 'out%d.jpg')).run()
+
+                thumbnails = os.listdir(folder)
+                thumbnails_media = []
+                for t, fname in enumerate(thumbnails):
+                    if fname[-3:] == 'jpg':
+                        thumbnails_media.append(Media(filename=os.path.join(folder, fname)).set("id", f"thumbnail_{t}"))
+                to_enrich.media[i].set("thumbnails", thumbnails_media)
--- a/src/auto_archiver/enrichers/wacz_enricher.py
+++ b/src/auto_archiver/enrichers/wacz_enricher.py
@@ -0,0 +1,100 @@
+import os, shutil, subprocess, uuid
+from loguru import logger
+
+from ..core import Media, Metadata, ArchivingContext
+from . import Enricher
+from ..utils import UrlUtil
+
+
+class WaczEnricher(Enricher):
+    """
+    Submits the current URL to the webarchive and returns a job_id or completed archive
+    """
+    name = "wacz_enricher"
+
+    def __init__(self, config: dict) -> None:
+        # without this STEP.__init__ is not called
+        super().__init__(config)
+
+    @staticmethod
+    def configs() -> dict:
+        return {
+            "profile": {"default": None, "help": "browsertrix-profile (for profile generation see https://github.com/webrecorder/browsertrix-crawler#creating-and-using-browser-profiles)."},
+            "timeout": {"default": 90, "help": "timeout for WACZ generation in seconds"},
+            "ignore_auth_wall": {"default": True, "help": "skip URL if it is behind authentication wall, set to False if you have browsertrix profile configured for private content."},
+        }
+
+    def enrich(self, to_enrich: Metadata) -> bool:
+        # TODO: figure out support for browsertrix in docker
+
+        url = to_enrich.get_url()
+
+        if UrlUtil.is_auth_wall(url):
+            logger.debug(f"[SKIP] SCREENSHOT since url is behind AUTH WALL: {url=}")
+            return
+        
+        collection = str(uuid.uuid4())[0:8]
+        browsertrix_home = os.path.abspath(ArchivingContext.get_tmp_dir())
+        
+        if os.getenv('RUNNING_IN_DOCKER'):
+            logger.debug(f"generating WACZ without Docker for {url=}")
+
+            cmd = [
+                "crawl",
+                "--url", url,
+                "--scopeType", "page",
+                "--generateWACZ",
+                "--text",
+                "--collection", collection,
+                "--id", collection,
+                "--saveState", "never",
+                "--behaviors", "autoscroll,autoplay,autofetch,siteSpecific",
+                "--behaviorTimeout", str(self.timeout),
+                "--timeout", str(self.timeout),
+                "--profile", str(self.profile)
+            ]
+        else:
+            logger.debug(f"generating WACZ in Docker for {url=}")
+            
+            cmd = [
+                "docker", "run",
+                "--rm",  # delete container once it has completed running
+                "-v", f"{browsertrix_home}:/crawls/",
+                # "-it", # this leads to "the input device is not a TTY"
+                "webrecorder/browsertrix-crawler", "crawl",
+                "--url", url,
+                "--scopeType", "page",
+                "--generateWACZ",
+                "--text",
+                "--collection", collection,
+                "--behaviors", "autoscroll,autoplay,autofetch,siteSpecific",
+                "--behaviorTimeout", str(self.timeout),
+                "--timeout", str(self.timeout)
+            ]
+
+            if self.profile:
+                profile_fn = os.path.join(browsertrix_home, "profile.tar.gz")
+                shutil.copyfile(self.profile, profile_fn)
+                # TODO: test which is right
+                cmd.extend(["--profile", profile_fn])
+                # cmd.extend(["--profile", "/crawls/profile.tar.gz"])
+
+        try:
+            logger.info(f"Running browsertrix-crawler: {' '.join(cmd)}")
+            subprocess.run(cmd, check=True)
+        except Exception as e:
+            logger.error(f"WACZ generation failed: {e}")
+            return False
+
+        
+
+        if os.getenv('RUNNING_IN_DOCKER'):
+            filename = os.path.join("collections", collection, f"{collection}.wacz")
+        else:
+            filename = os.path.join(browsertrix_home, "collections", collection, f"{collection}.wacz")
+        
+        if not os.path.exists(filename):
+            logger.warning(f"Unable to locate and upload WACZ  {filename=}")
+            return False
+
+        to_enrich.add_media(Media(filename), "browsertrix")
--- a/src/auto_archiver/enrichers/wayback_enricher.py
+++ b/src/auto_archiver/enrichers/wayback_enricher.py
@@ -0,0 +1,91 @@
+from loguru import logger
+import time, requests
+
+
+from . import Enricher
+from ..archivers import Archiver
+from ..utils import UrlUtil
+from ..core import Metadata
+
+class WaybackArchiverEnricher(Enricher, Archiver):
+    """
+    Submits the current URL to the webarchive and returns a job_id or completed archive
+    """
+    name = "wayback_archiver_enricher"
+
+    def __init__(self, config: dict) -> None:
+        # without this STEP.__init__ is not called
+        super().__init__(config)
+        assert type(self.secret) == str and len(self.secret) > 0, "please provide a value for the wayback_enricher API key"
+        assert type(self.secret) == str and len(self.secret) > 0, "please provide a value for the wayback_enricher API secret"
+
+    @staticmethod
+    def configs() -> dict:
+        return {
+            "timeout": {"default": 15, "help": "seconds to wait for successful archive confirmation from wayback, if more than this passes the result contains the job_id so the status can later be checked manually."},
+            "key": {"default": None, "help": "wayback API key. to get credentials visit https://archive.org/account/s3.php"},
+            "secret": {"default": None, "help": "wayback API secret. to get credentials visit https://archive.org/account/s3.php"}
+        }
+
+    def download(self, item: Metadata) -> Metadata:
+        result = Metadata()
+        result.merge(item)
+        if self.enrich(result):
+            return result.success("wayback")
+
+    def enrich(self, to_enrich: Metadata) -> bool:
+        url = to_enrich.get_url()
+        if UrlUtil.is_auth_wall(url):
+            logger.debug(f"[SKIP] WAYBACK since url is behind AUTH WALL: {url=}")
+            return
+
+        logger.debug(f"calling wayback for {url=}")
+
+        if to_enrich.get("wayback"):
+            logger.info(f"Wayback enricher had already been executed: {to_enrich.get('wayback')}")
+            return True
+
+        ia_headers = {
+            "Accept": "application/json",
+            "Authorization": f"LOW {self.key}:{self.secret}"
+        }
+        r = requests.post('https://web.archive.org/save/', headers=ia_headers, data={'url': url})
+
+        if r.status_code != 200:
+            logger.error(em := f"Internet archive failed with status of {r.status_code}: {r.json()}")
+            to_enrich.set("wayback", em)
+            return False
+
+        # check job status
+        job_id = r.json().get('job_id')
+        if not job_id:
+            logger.error(f"Wayback failed with {r.json()}")
+            return False
+
+        # waits at most timeout seconds until job is completed, otherwise only enriches the job_id information
+        start_time = time.time()
+        wayback_url = False
+        attempt = 1
+        while not wayback_url and time.time() - start_time <= self.timeout:
+            try:
+                logger.debug(f"GETting status for {job_id=} on {url=} ({attempt=})")
+                r_status = requests.get(f'https://web.archive.org/save/status/{job_id}', headers=ia_headers)
+                r_json = r_status.json()
+                if r_status.status_code == 200 and r_json['status'] == 'success':
+                    wayback_url = f"https://web.archive.org/web/{r_json['timestamp']}/{r_json['original_url']}"
+                elif r_status.status_code != 200 or r_json['status'] != 'pending':
+                    logger.error(f"Wayback failed with {r_json}")
+                    return False
+
+            except Exception as e:
+                logger.warning(f"error fetching status for {url=} due to: {e}")
+            if not wayback_url:
+                attempt += 1
+                time.sleep(1)  # TODO: can be improved with exponential backoff
+
+        if wayback_url:
+            to_enrich.set("wayback", wayback_url)
+        else:
+            to_enrich.set("wayback", {"job_id": job_id, "check_status": f'https://web.archive.org/save/status/{job_id}'})
+        to_enrich.set("check wayback", f"https://web.archive.org/web/*/{url}")
+        return True
--- a/src/auto_archiver/enrichers/whisper_enricher.py
+++ b/src/auto_archiver/enrichers/whisper_enricher.py
@@ -0,0 +1,130 @@
+import traceback
+import requests, time
+from loguru import logger
+
+from . import Enricher
+from ..core import Metadata, Media, ArchivingContext
+from ..storages import S3Storage
+
+
+class WhisperEnricher(Enricher):
+    """
+    Connects with a Whisper API service to get texts out of audio
+    whisper API repository: TODO
+    Only works if an S3 compatible storage is used
+    """
+    name = "whisper_enricher"
+
+    def __init__(self, config: dict) -> None:
+        # without this STEP.__init__ is not called
+        super().__init__(config)
+        assert type(self.api_key) == str and len(self.api_key) > 0, "please provide a value for the whisper_enricher api_key"
+        self.timeout = int(self.timeout)
+
+    @staticmethod
+    def configs() -> dict:
+        return {
+            "api_endpoint": {"default": "https://whisper.spoettel.dev/api/v1", "help": "WhisperApi api endpoint"},
+            "api_key": {"default": None, "help": "WhisperApi api key for authentication"},
+            "include_srt": {"default": False, "help": "Whether to include a subtitle SRT (SubRip Subtitle file) for the video (can be used in video players)."},
+            "timeout": {"default": 90, "help": "How many seconds to wait at most for a successful job completion."},
+            "action": {"default": "translation", "help": "which Whisper operation to execute", "choices": ["transcript", "translation", "language_detection"]},
+
+        }
+
+    def enrich(self, to_enrich: Metadata) -> None:
+        if not self._get_s3_storage():
+            logger.error("WhisperEnricher: To use the WhisperEnricher you need to use S3Storage so files are accessible publicly to the whisper service being called.")
+            return
+
+        url = to_enrich.get_url()
+        logger.debug(f"WHISPER[{self.action}]: iterating media items for {url=}.")
+
+        job_results = {}
+        for i, m in enumerate(to_enrich.media):
+            if m.is_video() or m.is_audio():
+                m.store(url=url)
+                try:
+                    job_id = self.submit_job(m)
+                    job_results[job_id] = False
+                    logger.debug(f"JOB SUBMITTED: {job_id=} for {m.key=}")
+                    to_enrich.media[i].set("whisper_model", {"job_id": job_id})
+                except Exception as e:
+                    logger.error(f"Failed to submit whisper job for {m.filename=} with error {e}\n{traceback.format_exc()}")
+
+        job_results = self.check_jobs(job_results)
+
+        for i, m in enumerate(to_enrich.media):
+            if m.is_video() or m.is_audio():
+                job_id = to_enrich.media[i].get("whisper_model")["job_id"]
+                to_enrich.media[i].set("whisper_model", {
+                    "job_id": job_id,
+                    **(job_results[job_id] if job_results[job_id] else {"result": "incomplete or failed job"})
+                })
+                # append the extracted text to the content of the post so it gets written to the DBs like gsheets text column
+                if job_results[job_id]:
+                    for k,v in job_results[job_id].items():
+                        if "_text" in k and len(v):
+                            to_enrich.set_content(f"\n[automatic video transcript]: {v}")
+
+    def submit_job(self, media: Media):
+        s3 = self._get_s3_storage()
+        s3_url = s3.get_cdn_url(media)
+        assert s3_url in media.urls, f"Could not find S3 url ({s3_url}) in list of stored media urls "
+        payload = {
+            "url": s3_url,
+            "type": self.action,
+            # "language": "string" # may be a config
+        }
+        response = requests.post(f'{self.api_endpoint}/jobs', json=payload, headers={'Authorization': f'Bearer {self.api_key}'})
+        assert response.status_code == 201, f"calling the whisper api {self.api_endpoint} returned a non-success code: {response.status_code}"
+        logger.debug(response.json())
+        return response.json()['id']
+
+    def check_jobs(self, job_results: dict):
+        start_time = time.time()
+        all_completed = False
+        while not all_completed and (time.time() - start_time) <= self.timeout:
+            all_completed = True
+            for job_id in job_results:
+                if job_results[job_id] != False: continue
+                all_completed = False  # at least one not ready
+                try: job_results[job_id] = self.check_job(job_id)
+                except Exception as e:
+                    logger.error(f"Failed to check {job_id=} with error {e}\n{traceback.format_exc()}")
+            if not all_completed: time.sleep(3)
+        return job_results
+
+    def check_job(self, job_id):
+        r = requests.get(f'{self.api_endpoint}/jobs/{job_id}', headers={'Authorization': f'Bearer {self.api_key}'})
+        assert r.status_code == 200, f"Job status did not respond with 200, instead with: {r.status_code}"
+        j = r.json()
+        logger.debug(f"Checked job {job_id=} with status='{j['status']}'")
+        if j['status'] == "processing": return False
+        elif j['status'] == "error": return f"Error: {j['meta']['error']}"
+        elif j['status'] == "success":
+            r_res = requests.get(f'{self.api_endpoint}/jobs/{job_id}/artifacts', headers={'Authorization': f'Bearer {self.api_key}'})
+            assert r_res.status_code == 200, f"Job artifacts did not respond with 200, instead with: {r_res.status_code}"
+            logger.success(r_res.json())
+            result = {}
+            for art_id, artifact in enumerate(r_res.json()):
+                subtitle = []
+                full_text = []
+                for i, d in enumerate(artifact.get("data")):
+                    subtitle.append(f"{i+1}\n{d.get('start')} --> {d.get('end')}\n{d.get('text').strip()}")
+                    full_text.append(d.get('text').strip())
+                if not len(subtitle): continue
+                if self.include_srt: result[f"artifact_{art_id}_subtitle"] = "\n".join(subtitle)
+                result[f"artifact_{art_id}_text"] = "\n".join(full_text)
+            # call /delete endpoint on timely success
+            r_del = requests.delete(f'{self.api_endpoint}/jobs/{job_id}', headers={'Authorization': f'Bearer {self.api_key}'})
+            logger.debug(f"DELETE whisper {job_id=} result: {r_del.status_code}")
+            return result
+        return False
+
+    def _get_s3_storage(self) -> S3Storage:
+        try:
+            return next(s for s in ArchivingContext.get("storages") if s.__class__ == S3Storage)
+        except:
+            logger.warning("No S3Storage instance found in storages")
+            return
--- a/src/auto_archiver/feeders/init.py
+++ b/src/auto_archiver/feeders/init.py
@@ -0,0 +1,3 @@
+from.feeder import Feeder
+from .gsheet_feeder import GsheetsFeeder
+from .cli_feeder import CLIFeeder
--- a/src/auto_archiver/feeders/cli_feeder.py
+++ b/src/auto_archiver/feeders/cli_feeder.py
@@ -0,0 +1,32 @@
+from loguru import logger
+
+from . import Feeder
+from ..core import Metadata, ArchivingContext
+
+
+class CLIFeeder(Feeder):
+    name = "cli_feeder"
+
+    def __init__(self, config: dict) -> None:
+        # without this STEP.__init__ is not called
+        super().__init__(config)
+        if type(self.urls) != list or len(self.urls) == 0:
+            raise Exception("CLI Feeder did not receive any URL to process")
+
+    @staticmethod
+    def configs() -> dict:
+        return {
+            "urls": {
+                "default": None,
+                "help": "URL(s) to archive, either a single URL or a list of urls, should not come from config.yaml",
+                "cli_set": lambda cli_val, cur_val: list(set(cli_val.split(",")))
+            },
+        }
+
+    def __iter__(self) -> Metadata:
+        for url in self.urls:
+            logger.debug(f"Processing {url}")
+            yield Metadata().set_url(url)
+            ArchivingContext.set("folder", "cli")
+
+        logger.success(f"Processed {len(self.urls)} URL(s)")
--- a/src/auto_archiver/feeders/feeder.py
+++ b/src/auto_archiver/feeders/feeder.py
@@ -0,0 +1,21 @@
+from __future__ import annotations
+from dataclasses import dataclass
+from abc import abstractmethod
+from ..core import Metadata
+from ..core import Step
+
+
+@dataclass
+class Feeder(Step):
+    name = "feeder"
+
+    def __init__(self, config: dict) -> None:
+        # without this STEP.__init__ is not called
+        super().__init__(config)
+
+    def init(name: str, config: dict) -> Feeder:
+        # only for code typing
+        return Step.init(name, config, Feeder)
+
+    @abstractmethod
+    def __iter__(self) -> Metadata: return None
--- a/src/auto_archiver/feeders/gsheet_feeder.py
+++ b/src/auto_archiver/feeders/gsheet_feeder.py
@@ -0,0 +1,92 @@
+import gspread, os
+
+from loguru import logger
+from slugify import slugify
+
+# from . import Enricher
+from . import Feeder
+from ..core import Metadata, ArchivingContext
+from ..utils import Gsheets, GWorksheet
+
+
+class GsheetsFeeder(Gsheets, Feeder):
+    name = "gsheet_feeder"
+
+    def __init__(self, config: dict) -> None:
+        # without this STEP.__init__ is not called
+        super().__init__(config)
+        self.gsheets_client = gspread.service_account(filename=self.service_account)
+
+    @staticmethod
+    def configs() -> dict:
+        return dict(
+            Gsheets.configs(),
+            ** {
+                "allow_worksheets": {
+                    "default": set(),
+                    "help": "(CSV) only worksheets whose name is included in allow are included (overrides worksheet_block), leave empty so all are allowed",
+                    "cli_set": lambda cli_val, cur_val: set(cli_val.split(","))
+                },
+                "block_worksheets": {
+                    "default": set(),
+                    "help": "(CSV) explicitly block some worksheets from being processed",
+                    "cli_set": lambda cli_val, cur_val: set(cli_val.split(","))
+                },
+                "use_sheet_names_in_stored_paths": {
+                    "default": True,
+                    "help": "if True the stored files path will include 'workbook_name/worksheet_name/...'",
+                }
+            })
+
+    def __iter__(self) -> Metadata:
+        sh = self.gsheets_client.open(self.sheet)
+        for ii, wks in enumerate(sh.worksheets()):
+            if not self.should_process_sheet(wks.title):
+                logger.debug(f"SKIPPED worksheet '{wks.title}' due to allow/block rules")
+                continue
+
+            logger.info(f'Opening worksheet {ii=}: {wks.title=} header={self.header}')
+            gw = GWorksheet(wks, header_row=self.header, columns=self.columns)
+
+            if len(missing_cols := self.missing_required_columns(gw)):
+                logger.warning(f"SKIPPED worksheet '{wks.title}' due to missing required column(s) for {missing_cols}")
+                continue
+
+            for row in range(1 + self.header, gw.count_rows() + 1):
+                url = gw.get_cell(row, 'url').strip()
+                if not len(url): continue
+
+                original_status = gw.get_cell(row, 'status')
+                status = gw.get_cell(row, 'status', fresh=original_status in ['', None])
+                # TODO: custom status parser(?) aka should_retry_from_status
+                if status not in ['', None]: continue
+
+                # All checks done - archival process starts here
+                m = Metadata().set_url(url)
+                ArchivingContext.set("gsheet", {"row": row, "worksheet": gw}, keep_on_reset=True)
+                folder = slugify(gw.get_cell(row, 'folder').strip())
+                if len(folder):
+                    if self.use_sheet_names_in_stored_paths:
+                        ArchivingContext.set("folder", os.path.join(folder, slugify(self.sheet), slugify(wks.title)), True)
+                    else:
+                        ArchivingContext.set("folder", folder, True)
+
+                yield m
+
+            logger.success(f'Finished worksheet {wks.title}')
+
+    def should_process_sheet(self, sheet_name: str) -> bool:
+        if len(self.allow_worksheets) and sheet_name not in self.allow_worksheets:
+            # ALLOW rules exist AND sheet name not explicitly allowed
+            return False
+        if len(self.block_worksheets) and sheet_name in self.block_worksheets:
+            # BLOCK rules exist AND sheet name is blocked
+            return False
+        return True
+
+    def missing_required_columns(self, gw: GWorksheet) -> list:
+        missing = []
+        for required_col in ['url', 'status']:
+            if not gw.col_exists(required_col):
+                missing.append(required_col)
+        return missing
--- a/src/auto_archiver/formatters/init.py
+++ b/src/auto_archiver/formatters/init.py
@@ -0,0 +1,3 @@
+from .formatter import Formatter
+from .html_formatter import HtmlFormatter
+from .mute_formatter import MuteFormatter
--- a/src/auto_archiver/formatters/formatter.py
+++ b/src/auto_archiver/formatters/formatter.py
@@ -0,0 +1,20 @@
+from __future__ import annotations
+from dataclasses import dataclass
+from abc import abstractmethod
+from ..core import Metadata, Media, Step
+
+
+@dataclass
+class Formatter(Step):
+    name = "formatter"
+
+    def __init__(self, config: dict) -> None:
+        # without this STEP.__init__ is not called
+        super().__init__(config)
+
+    def init(name: str, config: dict) -> Formatter:
+        # only for code typing
+        return Step.init(name, config, Formatter)
+
+    @abstractmethod
+    def format(self, item: Metadata) -> Media: return None
--- a/src/auto_archiver/formatters/html_formatter.py
+++ b/src/auto_archiver/formatters/html_formatter.py
@@ -0,0 +1,90 @@
+from __future__ import annotations
+from dataclasses import dataclass
+import mimetypes, uuid, os, pathlib
+from jinja2 import Environment, FileSystemLoader
+from urllib.parse import quote
+from loguru import logger
+
+from ..version import __version__
+from ..core import Metadata, Media, ArchivingContext
+from . import Formatter
+from ..enrichers import HashEnricher
+
+
+@dataclass
+class HtmlFormatter(Formatter):
+    name = "html_formatter"
+
+    def __init__(self, config: dict) -> None:
+        # without this STEP.__init__ is not called
+        super().__init__(config)
+        self.environment = Environment(loader=FileSystemLoader(os.path.join(pathlib.Path(__file__).parent.resolve(), "templates/")))
+        # JinjaHelper class static methods are added as filters
+        self.environment.filters.update({
+            k: v.__func__ for k, v in JinjaHelpers.__dict__.items() if isinstance(v, staticmethod)
+        })
+        self.template = self.environment.get_template("html_template.html")
+
+    @staticmethod
+    def configs() -> dict:
+        return {
+            "detect_thumbnails": {"default": True, "help": "if true will group by thumbnails generated by thumbnail enricher by id 'thumbnail_00'"}
+        }
+
+    def format(self, item: Metadata) -> Media:
+        url = item.get_url()
+        if item.is_empty():
+            logger.debug(f"[SKIP] FORMAT there is no media or metadata to format: {url=}")
+            return
+
+        content = self.template.render(
+            url=url,
+            title=item.get_title(),
+            media=item.media,
+            metadata=item.metadata,
+            version=__version__
+        )
+        html_path = os.path.join(ArchivingContext.get_tmp_dir(), f"formatted{str(uuid.uuid4())}.html")
+        with open(html_path, mode="w", encoding="utf-8") as outf:
+            outf.write(content)
+        final_media = Media(filename=html_path)
+
+        he = HashEnricher({"hash_enricher": {"algorithm": ArchivingContext.get("hash_enricher.algorithm"), "chunksize": 1.6e7}})
+        if len(hd := he.calculate_hash(final_media.filename)):
+            final_media.set("hash", f"{he.algorithm}:{hd}")
+
+        return final_media
+
+
+# JINJA helper filters
+class JinjaHelpers:
+    @staticmethod
+    def is_list(v) -> bool:
+        return isinstance(v, list)
+
+    @staticmethod
+    def is_video(s: str) -> bool:
+        m = mimetypes.guess_type(s)[0]
+        return "video" in (m or "")
+
+    @staticmethod
+    def is_image(s: str) -> bool:
+        m = mimetypes.guess_type(s)[0]
+        return "image" in (m or "")
+
+    @staticmethod
+    def is_audio(s: str) -> bool:
+        m = mimetypes.guess_type(s)[0]
+        return "audio" in (m or "")
+
+    @staticmethod
+    def is_media(v) -> bool:
+        return isinstance(v, Media)
+
+    @staticmethod
+    def get_extension(filename: str) -> str:
+        return os.path.splitext(filename)[1]
+
+    @staticmethod
+    def quote(s: str) -> str:
+        return quote(s)
--- a/src/auto_archiver/formatters/mute_formatter.py
+++ b/src/auto_archiver/formatters/mute_formatter.py
@@ -0,0 +1,16 @@
+from __future__ import annotations
+from dataclasses import dataclass
+
+from ..core import Metadata, Media
+from . import Formatter
+
+
+@dataclass
+class MuteFormatter(Formatter):
+    name = "mute_formatter"
+
+    def __init__(self, config: dict) -> None:
+        # without this STEP.__init__ is not called
+        super().__init__(config)
+
+    def format(self, item: Metadata) -> Media: return None
--- a/src/auto_archiver/formatters/templates/init.py
+++ b/src/auto_archiver/formatters/templates/init.py
--- a/src/auto_archiver/formatters/templates/html_template.html
+++ b/src/auto_archiver/formatters/templates/html_template.html
@@ -0,0 +1,217 @@
+{# templates/results.html #}
+{% import 'macros.html' as macros %}
+<!DOCTYPE html>
+<html lang="en">
+
+<head>
+    <meta charset="utf-8">
+    <link rel="stylesheet" href="https://fonts.googleapis.com/css?family=Roboto:300,300italic,700,700italic">
+    <title>{{ url }}</title>
+    <style>
+        html {
+            font-family: 'Roboto', sans-serif;
+        }
+
+        table {
+            table-layout: fixed;
+            width: 90%;
+        }
+
+        table td {
+            word-wrap: break-word;
+            overflow-wrap: break-word;
+            padding: 5px;
+        }
+
+        table,
+        th,
+        td {
+            margin: auto;
+            border: 1px solid;
+            border-collapse: collapse;
+            vertical-align: top;
+        }
+
+        table.metadata td:first-child {
+            text-align: center;
+        }
+
+        table.content td:nth-child(2),
+        .center {
+            text-align: center;
+        }
+
+        .copy:hover {
+            background: aliceblue;
+            cursor: copy;
+        }
+
+        #notification {
+            position: fixed;
+            right: 20px;
+            top: 20px;
+            background: aquamarine;
+            box-shadow: 6px 8px 5px 0px #000000;
+            padding: 10px;
+            font-size: large;
+            display: none;
+        }
+
+        img,
+        video {
+            filter: gray;
+            -webkit-filter: grayscale(1);
+            filter: grayscale(1);
+        }
+
+        /* Disable grayscale on hover */
+        img:hover,
+        video:hover {
+            -webkit-filter: grayscale(0);
+            filter: none;
+        }
+
+        .collapsible {
+            background-color: #777;
+            color: white;
+            cursor: pointer;
+            padding: 5px;
+            margin: 10px;
+            width: 100%;
+            border: none;
+            text-align: left;
+            outline: none;
+            font-size: 15px;
+        }
+
+        .active,
+        .collapsible:hover {
+            background-color: #555;
+        }
+
+        .collapsible-content {
+            padding: 0 18px;
+            display: none;
+            overflow: hidden;
+            background-color: #f1f1f1;
+        }
+    </style>
+</head>
+
+<body>
+    <div id="notification"></div>
+    <h2>Archived media for <a href="{{ url }}">{{ url }}</a></h2>
+    <p><b>title:</b> '<span class="copy">{{ title }}</span>'</p>
+    <h2 class="center">content {{ media | length }} item(s)</h2>
+    <table class="content">
+        <tr>
+            <th>about</th>
+            <th>preview(s)</th>
+        </tr>
+        {% for m in media %}
+        <tr>
+            <td>
+                <ul>
+                    <li><b>key:</b> <span class="copy">{{ m.key }}</span></li>
+                    <li><b>type:</b> <span class="copy">{{ m.mimetype }}</span></li>
+
+                    {% for prop in m.properties %}
+
+                    {% if m.properties[prop] | is_list %}
+                    <p></p>
+                    <div>
+                        <b class="collapsible" title="expand">{{ prop }}:</b>
+                        <p></p>
+                        <div class="collapsible-content">
+                            {% for subprop in m.properties[prop] %}
+                            {% if subprop | is_media %}
+                            {{ macros.display_media(subprop, false, url) }}
+                            {% else %}
+                            {{ subprop }}
+                            {% endif %}
+                            {% endfor %}
+                        </div>
+                    </div>
+                    <p></p>
+                    {% elif m.properties[prop] | string | length > 1 %}
+                    <li><b>{{ prop }}:</b> {{ macros.copy_urlize(m.properties[prop]) }}</li>
+                    {% endif %}
+
+                    {% endfor %}
+                </ul>
+            </td>
+            <td>
+                {{ macros.display_media(m, true, url) }}
+            </td>
+        </tr>
+        {% endfor %}
+    </table>
+    <h2 class="center">metadata</h2>
+    <table class="metadata">
+        <tr>
+            <th>key</th>
+            <th>value</th>
+        </tr>
+        {% for key in metadata %}
+        <tr>
+            <td>{{ key }}</td>
+            <td>
+                {{ macros.copy_urlize(metadata[key]) }}
+            </td>
+        </tr>
+        {% endfor %}
+    </table>
+
+    <p style="text-align:center;">Made with <a href="https://github.com/bellingcat/auto-archiver">bellingcat/auto-archiver</a> v{{ version }}</p>
+</body>
+<script defer>
+    // notification logic
+    const notification = document.getElementById("notification");
+
+    function showNotification(message, miliseconds) {
+        notification.style.display = "block";
+        notification.innerText = message;
+        setTimeout(() => {
+            notification.style.display = "none";
+            notification.innerText = "";
+        }, miliseconds || 1000)
+    }
+
+    // copy logic
+    Array.from(document.querySelectorAll(".copy")).forEach(el => {
+        el.onclick = () => {
+            document.execCommand("copy");
+        }
+        el.addEventListener("copy", (e) => {
+            e.preventDefault();
+            if (e.clipboardData) {
+                if (el.hasAttribute("copy-value")) {
+                    e.clipboardData.setData("text/plain", el.getAttribute("copy-value"));
+                } else {
+                    e.clipboardData.setData("text/plain", el.textContent);
+                }
+                console.log(e.clipboardData.getData("text"))
+                showNotification("copied!")
+            }
+        })
+    })
+
+    // collapsibles
+    let coll = document.getElementsByClassName("collapsible");
+    let i;
+
+    for (i = 0; i < coll.length; i++) {
+        coll[i].addEventListener("click", function() {
+            this.classList.toggle("active");
+            // let content = this.nextElementSibling;
+            let content = this.parentElement.querySelector(".collapsible-content");
+            if (content.style.display === "block") {
+                content.style.display = "none";
+            } else {
+                content.style.display = "block";
+            }
+        });
+    }
+</script>
+
+</html>
--- a/src/auto_archiver/formatters/templates/macros.html
+++ b/src/auto_archiver/formatters/templates/macros.html
@@ -0,0 +1,77 @@
+{% macro display_media(m, links, main_url) -%}
+
+{% for url in m.urls %}
+{% if url | length == 0 %}
+No URL available for {{ m.key }}.
+{% elif 'http' in url %}
+{% if 'image' in m.mimetype %}
+<div>
+    <a href="{{ url }}">
+        <img src="{{ url }}" style="max-height:400px;max-width:400px;"></img>
+    </a>
+
+    <div>
+        Reverse Image Search:&nbsp;
+        <a href="https://www.google.com/searchbyimage?sbisrc=4chanx&image_url={{ url | quote }}&safe=off">Google</a>,&nbsp;
+        <a href="https://lens.google.com/uploadbyurl?url={{ url | quote }}">Google Lens</a>,&nbsp;
+        <a href="https://yandex.ru/images/touch/search?rpt=imageview&url={{ url | quote }}">Yandex</a>,&nbsp;
+        <a href="https://www.bing.com/images/search?view=detailv2&iss=sbi&form=SBIVSP&sbisrc=UrlPaste&q=imgurl:{{ url | quote }}">Bing</a>,&nbsp;
+        <a href="https://www.tineye.com/search/?url={{ url | quote }}">Tineye</a>,&nbsp;
+        <a href="https://iqdb.org/?url={{ url | quote }}">IQDB</a>,&nbsp;
+        <a href="https://saucenao.com/search.php?db=999&url={{ url | quote }}">SauceNAO</a>,&nbsp;
+        <a href="https://imgops.com/{{ url | quote }}">IMGOPS</a>
+    </div>
+    <p></p>
+</div>
+{% elif 'video' in m.mimetype %}
+<div>
+    <video src="{{ url }}" controls style="max-height:400px;max-width:600px;">
+        Your browser does not support the video element.
+    </video>
+</div>
+{% elif 'audio' in m.mimetype %}
+<div>
+    <audio controls>
+        <source src="{{ url }}" type="{{ m.mimetype }}">
+        Your browser does not support the audio element.
+    </audio>
+</div>
+{% elif m.filename | get_extension == ".wacz" %}
+<a href="https://replayweb.page/?source={{ url | quote }}#view=pages&url={{ main_url }}">replayweb</a>
+{% else %}
+No preview available for {{ m.key }}.
+{% endif %}
+{% else %}
+{{ m.url | urlize }}
+{% endif %}
+{% if links %}
+<a href="{{ url }}">open</a> or
+<a href="{{ url }}" download="">download</a> or
+{{ copy_urlize(url, "copy") }}
+
+<br>
+{% endif %}
+{% endfor %}
+
+{%- endmacro -%}
+
+{% macro copy_urlize(val, href_text) -%}
+
+{% if val is mapping %}
+<ul>
+    {% for key in val %}
+    <li>
+        <b>{{ key }}:</b> {{ copy_urlize(val[key]) }}
+    </li>
+    {% endfor %}
+</ul>
+
+{% else %}
+{% if href_text  | length == 0 %}
+<span class="copy">{{ val | string | urlize }}</span>
+{% else %}
+<span class="copy" copy-value="{{val}}">{{ href_text | string | urlize }}</span>
+{% endif %}
+{% endif %}
+
+{%- endmacro -%}
--- a/src/auto_archiver/storages/init.py
+++ b/src/auto_archiver/storages/init.py
@@ -0,0 +1,4 @@
+from .storage import Storage
+from .s3 import S3Storage
+from .local import LocalStorage
+from .gd import GDriveStorage
--- a/src/auto_archiver/storages/gd.py
+++ b/src/auto_archiver/storages/gd.py
@@ -1,32 +1,27 @@
-import os, time

+import shutil, os, time, json
+from typing import IO
 from loguru import logger
-from .base_storage import Storage
-from dataclasses import dataclass
+
 from googleapiclient.discovery import build
 from googleapiclient.http import MediaFileUpload
 from google.oauth2 import service_account
-
-
 from google.oauth2.credentials import Credentials
 from google.auth.transport.requests import Request

-@dataclass
-class GDConfig:
-    root_folder_id: str
-    oauth_token_filename: str
-    service_account: str = "service_account.json"
-    folder: str = "default"
+from ..core import Media
+from . import Storage

-class GDStorage(Storage):
-    def __init__(self, config: GDConfig):
-        self.folder = config.folder
-        self.root_folder_id = config.root_folder_id
-        
-        SCOPES=['https://www.googleapis.com/auth/drive']
-        
-        token_file = config.oauth_token_filename
-        if token_file is not None:
+
+class GDriveStorage(Storage):
+    name = "gdrive_storage"
+
+    def __init__(self, config: dict) -> None:
+        super().__init__(config)
+
+        SCOPES = ['https://www.googleapis.com/auth/drive']
+
+        if self.oauth_token is not None:
            """
            Tokens are refreshed after 1 hour 
            however keep working for 7 days (tbc)
@@ -35,8 +30,13 @@ class GDStorage(Storage):
            see this link for details on the token
            https://davemateer.com/2022/04/28/google-drive-with-python#tokens
            """
-            logger.debug(f'Using GD OAuth token {token_file}')
-            creds = Credentials.from_authorized_user_file(token_file, SCOPES)
+            logger.debug(f'Using GD OAuth token {self.oauth_token}')
+            # workaround for missing 'refresh_token' in from_authorized_user_file
+            with open(self.oauth_token, 'r') as stream:
+                creds_json = json.load(stream)
+                creds_json['refresh_token'] = creds_json.get("refresh_token", "")
+            creds = Credentials.from_authorized_user_info(creds_json, SCOPES)
+            # creds = Credentials.from_authorized_user_file(self.oauth_token, SCOPES)

            if not creds or not creds.valid:
                if creds and creds.expired and creds.refresh_token:
@@ -46,7 +46,7 @@ class GDStorage(Storage):
                    raise Exception("Problem with creds - create the token again")

                # Save the credentials for the next run
-                with open(token_file, 'w') as token:
+                with open(self.oauth_token, 'w') as token:
                    logger.debug('Saving new GD OAuth token')
                    token.write(creds.to_json())
            else:
@@ -58,18 +58,27 @@ class GDStorage(Storage):

        self.service = build('drive', 'v3', credentials=creds)

-    def get_cdn_url(self, key):
+    @staticmethod
+    def configs() -> dict:
+        return dict(
+            Storage.configs(),
+            ** {
+                "root_folder_id": {"default": None, "help": "root google drive folder ID to use as storage, found in URL: 'https://drive.google.com/drive/folders/FOLDER_ID'"},
+                "oauth_token": {"default": None, "help": "JSON filename with Google Drive OAuth token: check auto-archiver repository scripts folder for create_update_gdrive_oauth_token.py. NOTE: storage used will count towards owner of GDrive folder, therefore it is best to use oauth_token_filename over service_account."},
+                "service_account": {"default": "secrets/service_account.json", "help": "service account JSON file path, same as used for Google Sheets. NOTE: storage used will count towards the developer account."},
+            })
+
+    def get_cdn_url(self, media: Media) -> str:
        """
        only support files saved in a folder for GD
        S3 supports folder and all stored in the root
        """
-        key = self.clean_key(key)

-        full_name = os.path.join(self.folder, key)
+        # full_name = os.path.join(self.folder, media.key)
        parent_id, folder_id = self.root_folder_id, None
-        path_parts = full_name.split(os.path.sep)
+        path_parts = media.key.split(os.path.sep)
        filename = path_parts[-1]
-        logger.info(f"looking for folders for {path_parts[0:-1]} before uploading {filename=}")
+        logger.info(f"looking for folders for {path_parts[0:-1]} before getting url for {filename=}")
        for folder in path_parts[0:-1]:
            folder_id = self._get_id_from_parent_and_name(parent_id, folder, use_mime_type=True, raise_on_missing=True)
            parent_id = folder_id
@@ -78,22 +87,23 @@ class GDStorage(Storage):
        file_id = self._get_id_from_parent_and_name(folder_id, filename)
        return f"https://drive.google.com/file/d/{file_id}/view?usp=sharing"

-    def exists(self, key):
-        try:
-            self.get_cdn_url(key)
-            return True
-        except: return False
+    def upload(self, media: Media, **kwargs) -> bool:
+        # override parent so that we can use shutil.copy2 and keep metadata
+        dest = os.path.join(self.save_to, media.key)
+        os.makedirs(os.path.dirname(dest), exist_ok=True)
+        logger.debug(f'[{self.__class__.name}] storing file {media.filename} with key {media.key} to {dest}')
+        res = shutil.copy2(media.filename, dest)
+        logger.info(res)
+        return True

-    def uploadf(self, file: str, key: str, **_kwargs):
+    def upload(self, media: Media, **kwargs) -> bool:
+        logger.debug(f'[{self.__class__.name}] storing file {media.filename} with key {media.key}')
        """
        1. for each sub-folder in the path check if exists or create
        2. upload file to root_id/other_paths.../filename
        """
-        key = self.clean_key(key)
-
-        full_name = os.path.join(self.folder, key)
        parent_id, upload_to = self.root_folder_id, None
-        path_parts = full_name.split(os.path.sep)
+        path_parts = media.key.split(os.path.sep)
        filename = path_parts[-1]
        logger.info(f"checking folders {path_parts[0:-1]} exist (or creating) before uploading {filename=}")
        for folder in path_parts[0:-1]:
@@ -108,22 +118,13 @@ class GDStorage(Storage):
            'name': [filename],
            'parents': [upload_to]
        }
-        media = MediaFileUpload(file, resumable=True)
+        media = MediaFileUpload(media.filename, resumable=True)
        gd_file = self.service.files().create(body=file_metadata, media_body=media, fields='id').execute()
-        logger.debug(f'uploadf: uploaded file {gd_file["id"]} succesfully in folder={upload_to}')
+        logger.debug(f'uploadf: uploaded file {gd_file["id"]} successfully in folder={upload_to}')

-    def upload(self, filename: str, key: str, **kwargs):
-        # GD only requires the filename not a file reader
-        self.uploadf(filename, key, **kwargs)
+    # must be implemented even if unused
+    def uploadf(self, file: IO[bytes], key: str, **kwargs: dict) -> bool: pass

-    def clean_key(self, key):
-        # GDrive does not work well with trailing forward slashes and some keys come with that
-        if key.startswith('/'):
-            logger.debug(f'Found and fixed a leading "/" for {key=}')
-            return key[1:]
-        return key
-
-    # gets the Drive folderID if it is there
    def _get_id_from_parent_and_name(self, parent_id: str, name: str, retries: int = 1, sleep_seconds: int = 10, use_mime_type: bool = False, raise_on_missing: bool = True, use_cache=False):
        """
        Retrieves the id of a folder or file from its @name and the @parent_id folder
@@ -183,3 +184,9 @@ class GDStorage(Storage):
        }
        gd_folder = self.service.files().create(body=file_metadata, fields='id').execute()
        return gd_folder.get('id')
+
+    # def exists(self, key):
+    #     try:
+    #         self.get_cdn_url(key)
+    #         return True
+    #     except: return False
--- a/src/auto_archiver/storages/local.py
+++ b/src/auto_archiver/storages/local.py
@@ -0,0 +1,44 @@
+
+import shutil
+from typing import IO
+import os
+from loguru import logger
+
+from ..core import Media
+from ..storages import Storage
+
+
+class LocalStorage(Storage):
+    name = "local_storage"
+
+    def __init__(self, config: dict) -> None:
+        super().__init__(config)
+        os.makedirs(self.save_to, exist_ok=True)
+
+    @staticmethod
+    def configs() -> dict:
+        return dict(
+            Storage.configs(),
+            ** {
+                "save_to": {"default": "./archived", "help": "folder where to save archived content"},
+                "save_absolute": {"default": False, "help": "whether the path to the stored file is absolute or relative in the output result inc. formatters (WARN: leaks the file structure)"},
+            })
+
+    def get_cdn_url(self, media: Media) -> str:
+        # TODO: is this viable with Storage.configs on path/filename?
+        dest = os.path.join(self.save_to, media.key)
+        if self.save_absolute:
+            dest = os.path.abspath(dest)
+        return dest
+
+    def upload(self, media: Media, **kwargs) -> bool:
+        # override parent so that we can use shutil.copy2 and keep metadata
+        dest = os.path.join(self.save_to, media.key)
+        os.makedirs(os.path.dirname(dest), exist_ok=True)
+        logger.debug(f'[{self.__class__.name}] storing file {media.filename} with key {media.key} to {dest}')
+        res = shutil.copy2(media.filename, dest)
+        logger.info(res)
+        return True
+
+    # must be implemented even if unused
+    def uploadf(self, file: IO[bytes], key: str, **kwargs: dict) -> bool: pass
--- a/src/auto_archiver/storages/s3.py
+++ b/src/auto_archiver/storages/s3.py
@@ -0,0 +1,73 @@
+
+from typing import IO, Any
+import boto3, uuid, os, mimetypes
+from botocore.errorfactory import ClientError
+from ..core import Metadata
+from ..core import Media
+from ..storages import Storage
+from loguru import logger
+from slugify import slugify
+
+
+class S3Storage(Storage):
+    name = "s3_storage"
+
+    def __init__(self, config: dict) -> None:
+        super().__init__(config)
+        self.s3 = boto3.client(
+            's3',
+            region_name=self.region,
+            endpoint_url=self.endpoint_url.format(region=self.region),
+            aws_access_key_id=self.key,
+            aws_secret_access_key=self.secret
+        )
+
+    @staticmethod
+    def configs() -> dict:
+        return dict(
+            Storage.configs(),
+            ** {
+                "bucket": {"default": None, "help": "S3 bucket name"},
+                "region": {"default": None, "help": "S3 region name"},
+                "key": {"default": None, "help": "S3 API key"},
+                "secret": {"default": None, "help": "S3 API secret"},
+                # TODO: how to have sth like a custom folder? has to come from the feeders
+                "endpoint_url": {
+                    "default": 'https://{region}.digitaloceanspaces.com',
+                    "help": "S3 bucket endpoint, {region} are inserted at runtime"
+                },
+                "cdn_url": {
+                    "default": 'https://{bucket}.{region}.cdn.digitaloceanspaces.com/{key}',
+                    "help": "S3 CDN url, {bucket}, {region} and {key} are inserted at runtime"
+                },
+                "private": {"default": False, "help": "if true S3 files will not be readable online"},
+            })
+
+    def get_cdn_url(self, media: Media) -> str:
+        return self.cdn_url.format(bucket=self.bucket, region=self.region, key=media.key)
+
+    def uploadf(self, file: IO[bytes], media: Media, **kwargs: dict) -> None:
+        extra_args = kwargs.get("extra_args", {})
+        if not self.private and 'ACL' not in extra_args:
+            extra_args['ACL'] = 'public-read'
+
+        if 'ContentType' not in extra_args:
+            try:
+                if media.mimetype:
+                    extra_args['ContentType'] = media.mimetype
+            except Exception as e:
+                logger.warning(f"Unable to get mimetype for {media.key=}, error: {e}")
+
+        self.s3.upload_fileobj(file, Bucket=self.bucket, Key=media.key, ExtraArgs=extra_args)
+        return True
+
+    # def exists(self, key: str) -> bool:
+    #     """
+    #     Tests if a given file with key=key exists in the bucket
+    #     """
+    #     try:
+    #         self.s3.head_object(Bucket=self.bucket, Key=key)
+    #         return True
+    #     except ClientError as e:
+    #         logger.warning(f"got a ClientError when checking if {key=} exists in bucket={self.bucket}: {e}")
+    #     return False
--- a/src/auto_archiver/storages/storage.py
+++ b/src/auto_archiver/storages/storage.py
@@ -0,0 +1,84 @@
+from __future__ import annotations
+from abc import abstractmethod
+from dataclasses import dataclass
+from typing import IO
+
+from ..core import Media, Step, ArchivingContext
+from ..enrichers import HashEnricher
+from loguru import logger
+import os, uuid
+from slugify import slugify
+
+
+@dataclass
+class Storage(Step):
+    name = "storage"
+    PATH_GENERATOR_OPTIONS = ["flat", "url", "random"]
+    FILENAME_GENERATOR_CHOICES = ["random", "static"]
+
+    def __init__(self, config: dict) -> None:
+        # without this STEP.__init__ is not called
+        super().__init__(config)
+        assert self.path_generator in Storage.PATH_GENERATOR_OPTIONS, f"path_generator must be one of {Storage.PATH_GENERATOR_OPTIONS}"
+        assert self.filename_generator in Storage.FILENAME_GENERATOR_CHOICES, f"filename_generator must be one of {Storage.FILENAME_GENERATOR_CHOICES}"
+
+    @staticmethod
+    def configs() -> dict:
+        return {
+            "path_generator": {
+                "default": "url",
+                "help": "how to store the file in terms of directory structure: 'flat' sets to root; 'url' creates a directory based on the provided URL; 'random' creates a random directory.",
+                "choices": Storage.PATH_GENERATOR_OPTIONS
+            },
+            "filename_generator": {
+                "default": "random",
+                "help": "how to name stored files: 'random' creates a random string; 'static' uses a replicable strategy such as a hash.",
+                "choices": Storage.FILENAME_GENERATOR_CHOICES
+            }
+        }
+
+    def init(name: str, config: dict) -> Storage:
+        # only for typing...
+        return Step.init(name, config, Storage)
+
+    def store(self, media: Media, url: str) -> None:
+        if media.is_stored(): 
+            logger.debug(f"{media.key} already stored, skipping")
+            return
+        self.set_key(media, url)
+        self.upload(media)
+        media.add_url(self.get_cdn_url(media))
+
+    @abstractmethod
+    def get_cdn_url(self, media: Media) -> str: pass
+
+    @abstractmethod
+    def uploadf(self, file: IO[bytes], key: str, **kwargs: dict) -> bool: pass
+
+    def upload(self, media: Media, **kwargs) -> bool:
+        logger.debug(f'[{self.__class__.name}] storing file {media.filename} with key {media.key}')
+        with open(media.filename, 'rb') as f:
+            return self.uploadf(f, media, **kwargs)
+
+    def set_key(self, media: Media, url) -> None:
+        """takes the media and optionally item info and generates a key"""
+        if media.key is not None and len(media.key) > 0: return
+        folder = ArchivingContext.get("folder", "")
+        filename, ext = os.path.splitext(media.filename)
+
+        # path_generator logic
+        if self.path_generator == "flat":
+            path = ""
+            filename = slugify(filename)  # in case it comes with os.sep
+        elif self.path_generator == "url": path = slugify(url)
+        elif self.path_generator == "random":
+            path = ArchivingContext.get("random_path", str(uuid.uuid4())[:16], True)
+
+        # filename_generator logic
+        if self.filename_generator == "random": filename = str(uuid.uuid4())[:16]
+        elif self.filename_generator == "static":
+            he = HashEnricher({"hash_enricher": {"algorithm": ArchivingContext.get("hash_enricher.algorithm"), "chunksize": 1.6e7}})
+            hd = he.calculate_hash(media.filename)
+            filename = hd[:24]
+
+        media.key = os.path.join(folder, path, f"{filename}{ext}")
--- a/src/auto_archiver/utils/init.py
+++ b/src/auto_archiver/utils/init.py
@@ -0,0 +1,6 @@
+# we need to explicitly expose the available imports here
+from .gworksheet import GWorksheet
+from .misc import *
+from .webdriver import Webdriver
+from .gsheet import Gsheets
+from .url import UrlUtil
--- a/src/auto_archiver/utils/gsheet.py
+++ b/src/auto_archiver/utils/gsheet.py
@@ -0,0 +1,44 @@
+import json, gspread
+
+from ..core import Step
+
+
+class Gsheets(Step):
+    name = "gsheets"
+
+    def __init__(self, config: dict) -> None:
+        # without this STEP.__init__ is not called
+        super().__init__(config)
+        self.gsheets_client = gspread.service_account(filename=self.service_account)
+        #TODO: config should be responsible for conversions
+        try: self.header = int(self.header)
+        except: pass
+        assert type(self.header) == int, f"header ({self.header}) value must be an integer not {type(self.header)}"
+        assert self.sheet is not None, "You need to define a sheet name in your orchestration file when using gsheets."
+
+    @staticmethod
+    def configs() -> dict:
+        return {
+            "sheet": {"default": None, "help": "name of the sheet to archive"},
+            "header": {"default": 1, "help": "index of the header row (starts at 1)"},
+            "service_account": {"default": "secrets/service_account.json", "help": "service account JSON file path"},
+            "columns": {
+                "default": {
+                    'url': 'link',
+                    'status': 'archive status',
+                    'folder': 'destination folder',
+                    'archive': 'archive location',
+                    'date': 'archive date',
+                    'thumbnail': 'thumbnail',
+                    'timestamp': 'upload timestamp',
+                    'title': 'upload title',
+                    'text': 'text content',
+                    'screenshot': 'screenshot',
+                    'hash': 'hash',
+                    'wacz': 'wacz',
+                    'replaywebpage': 'replaywebpage',
+                },
+                "help": "names of columns in the google sheet (stringified JSON object)",
+                "cli_set": lambda cli_val, cur_val: dict(cur_val, **json.loads(cli_val))
+            },
+        }
--- a/src/auto_archiver/utils/gworksheet.py
+++ b/src/auto_archiver/utils/gworksheet.py
@@ -15,10 +15,8 @@ class GWorksheet:
        'archive': 'archive location',
        'date': 'archive date',
        'thumbnail': 'thumbnail',
-        'thumbnail_index': 'thumbnail index',
        'timestamp': 'upload timestamp',
        'title': 'upload title',
-        'duration': 'duration',
        'screenshot': 'screenshot',
        'hash': 'hash',
        'wacz': 'wacz',
@@ -40,11 +38,11 @@ class GWorksheet:

    def _col_index(self, col: str):
        self._check_col_exists(col)
-        return self.headers.index(self.columns[col])
+        return self.headers.index(self.columns[col].lower())

    def col_exists(self, col: str):
        self._check_col_exists(col)
-        return self.columns[col] in self.headers
+        return self.columns[col].lower() in self.headers

    def count_rows(self):
        return len(self.values)
@@ -98,7 +96,7 @@ class GWorksheet:
        cell_updates = [
            {
                'range': self.to_a1(row, col),
-                'values': [[val]]
+                'values': [[str(val)[0:49999]]]
            }
            for row, col, val in cell_updates
        ]
--- a/src/auto_archiver/utils/misc.py
+++ b/src/auto_archiver/utils/misc.py
@@ -29,3 +29,14 @@ def getattr_or(o: object, prop: str, default=None):
    except:
        return default

+
+class DateTimeEncoder(json.JSONEncoder):
+    # to allow json.dump with datetimes do json.dumps(obj, cls=DateTimeEncoder)
+    def default(self, o):
+        if isinstance(o, datetime):
+            return str(o)  # with timezone
+        return json.JSONEncoder.default(self, o)
+
+
+def dump_payload(p):
+    return json.dumps(p, ensure_ascii=False, indent=4, cls=DateTimeEncoder)
--- a/src/auto_archiver/utils/url.py
+++ b/src/auto_archiver/utils/url.py
@@ -0,0 +1,19 @@
+import re
+
+class UrlUtil:
+    telegram_private = re.compile(r"https:\/\/t\.me(\/c)\/(.+)\/(\d+)")
+    is_istagram = re.compile(r"https:\/\/www\.instagram\.com")
+
+    @staticmethod
+    def clean(url): return url
+
+    @staticmethod
+    def is_auth_wall(url):
+        """
+        checks if URL is behind an authentication wall meaning steps like wayback, wacz, ... may not work
+        """
+        if UrlUtil.telegram_private.match(url): return True
+        if UrlUtil.is_istagram.match(url): return True
+
+        return False
+
--- a/src/auto_archiver/utils/webdriver.py
+++ b/src/auto_archiver/utils/webdriver.py
@@ -0,0 +1,45 @@
+from __future__ import annotations
+from selenium import webdriver
+from selenium.common.exceptions import TimeoutException
+from loguru import logger
+from selenium.webdriver.common.by import By
+import time
+
+
+class Webdriver:
+    def __init__(self, width: int, height: int, timeout_seconds: int, facebook_accept_cookies: bool = False) -> webdriver:
+        self.width = width
+        self.height = height
+        self.timeout_seconds = timeout_seconds
+        self.facebook_accept_cookies = facebook_accept_cookies
+
+    def __enter__(self) -> webdriver:
+        options = webdriver.FirefoxOptions()
+        options.headless = True
+        options.set_preference('network.protocol-handler.external.tg', False)
+        try:
+            self.driver = webdriver.Firefox(options=options)
+            self.driver.set_window_size(self.width, self.height)
+            self.driver.set_page_load_timeout(self.timeout_seconds)
+        except TimeoutException as e:
+            logger.error(f"failed to get new webdriver, possibly due to insufficient system resources or timeout settings: {e}")
+
+        if self.facebook_accept_cookies:
+            try:
+                logger.debug(f'Trying fb click accept cookie popup.')
+                self.driver.get("http://www.facebook.com")
+                foo = self.driver.find_element(By.XPATH, "//button[@data-cookiebanner='accept_only_essential_button']")
+                foo.click()
+                logger.debug(f'fb click worked')
+                # linux server needs a sleep otherwise facebook cookie won't have worked and we'll get a popup on next page
+                time.sleep(2)
+            except:
+                logger.warning(f'Failed on fb accept cookies.')
+
+        return self.driver
+
+    def __exit__(self, exc_type, exc_val, exc_tb):
+        self.driver.close()
+        self.driver.quit()
+        del self.driver
+        return True
--- a/src/auto_archiver/version.py
+++ b/src/auto_archiver/version.py
@@ -0,0 +1,12 @@
+
+_MAJOR = "0"
+_MINOR = "5"
+# On main and in a nightly release the patch should be one ahead of the last
+# released build.
+_PATCH = "12"
+# This is mainly for nightly builds which have the suffix ".dev$DATE". See
+# https://semver.org/#is-v123-a-semantic-version for the semantics.
+_SUFFIX = ""
+
+VERSION_SHORT = "{0}.{1}".format(_MAJOR, _MINOR)
+__version__ = "{0}.{1}.{2}{3}".format(_MAJOR, _MINOR, _PATCH, _SUFFIX)
--- a/storages/init.py
+++ b/storages/init.py
@@ -1,5 +0,0 @@
-# we need to explicitly expose the available imports here
-from .base_storage import Storage
-from .local_storage import LocalStorage, LocalConfig
-from .s3_storage import S3Config, S3Storage
-from .gd_storage import GDConfig, GDStorage
--- a/storages/base_storage.py
+++ b/storages/base_storage.py
@@ -1,24 +0,0 @@
-from loguru import logger
-from abc import ABC, abstractmethod
-from pathlib import Path
-
-
-class Storage(ABC):
-    TMP_FOLDER = "tmp/"
-
-    @abstractmethod
-    def __init__(self, config): pass
-
-    @abstractmethod
-    def get_cdn_url(self, key): pass
-
-    @abstractmethod
-    def exists(self, key): pass
-
-    @abstractmethod
-    def uploadf(self, file, key, **kwargs): pass
-
-    def upload(self, filename: str, key: str, **kwargs):
-        logger.debug(f'[{self.__class__.__name__}] uploading file {filename} with key {key}')
-        with open(filename, 'rb') as f:
-            self.uploadf(f, key, **kwargs)
--- a/storages/local_storage.py
+++ b/storages/local_storage.py
@@ -1,31 +0,0 @@
-import os
-
-from dataclasses import dataclass
-
-from .base_storage import Storage
-from utils import mkdir_if_not_exists
-
-
-@dataclass
-class LocalConfig:
-    folder: str = ""
-    save_to: str = "./"
-
-class LocalStorage(Storage):
-    def __init__(self, config:LocalConfig):
-        self.folder = config.folder
-        self.save_to = config.save_to
-        mkdir_if_not_exists(self.save_to)
-
-    def get_cdn_url(self, key):
-        full_path = os.path.join(self.save_to, self.folder, key)
-        mkdir_if_not_exists(os.path.join(*full_path.split(os.path.sep)[0:-1]))
-        return os.path.abspath(full_path)
-
-    def exists(self, key):
-        return os.path.isfile(self.get_cdn_url(key))
-
-    def uploadf(self, file, key, **kwargs):
-        path = self.get_cdn_url(key)
-        with open(path, "wb") as outf:
-            outf.write(file.read())
--- a/Show More
+++ b/Show More