enables option to toggle db api writes

2026-06-09 11:58:28 +03:00 · 2023-12-13 12:54:12 +00:00
308 changed files with 5504 additions and 23628 deletions
--- a/.github/workflows/docker-publish.yaml
+++ b/.github/workflows/docker-publish.yaml
@@ -8,10 +8,13 @@ name: Docker
 on:
  release:
    types: [published]
+  push:
+    # branches: [ "main" ]
+    tags: [ "v*.*.*" ]

 env:
  # Use docker.io for Docker Hub if empty
-  REGISTRY: docker.io
+  REGISTRY: ghcr.io
  # github.repository as <account>/<repo>
  IMAGE_NAME: ${{ github.repository }}

@@ -25,32 +28,30 @@ jobs:
        uses: actions/checkout@v3
        
      - name: Set up QEMU
-        uses: docker/setup-qemu-action@v3
+        uses: docker/setup-qemu-action@v1
      # https://github.com/docker/setup-buildx-action
-
+      
      - name: Set up Docker Buildx
        id: buildx
-        uses: docker/setup-buildx-action@v3
-
+        uses: docker/setup-buildx-action@v1
+      
      - name: Log in to Docker Hub
-        uses: docker/login-action@9780b0c442fbb1117ed29e0efdff1e18412f7567
+        uses: docker/login-action@f054a8b539a109f9f41c372932f1ae047eff08c9
        with:
          username: ${{ secrets.DOCKER_USERNAME }}
          password: ${{ secrets.DOCKER_PASSWORD }}
-
+      
      - name: Extract metadata (tags, labels) for Docker
        id: meta
-        uses: docker/metadata-action@369eb591f429131d6889c46b94e711f089e6ca96
+        uses: docker/metadata-action@98669ae865ea3cffbcbaa878cf57c20bbf1c6c38
        with:
          images: bellingcat/auto-archiver
      
      - name: Build and push Docker image
-        uses: docker/build-push-action@v6
+        uses: docker/build-push-action@v2
        with:
          context: .
          platforms: linux/amd64,linux/arm64
          push: ${{ github.event_name != 'pull_request' }}
          tags: ${{ steps.meta.outputs.tags }}
          labels: ${{ steps.meta.outputs.labels }}
-          cache-from: type=registry,ref=${{ env.REGISTRY }}/${{ env.IMAGE_NAME }}:cache
-          cache-to: type=registry,ref=${{ env.REGISTRY }}/${{ env.IMAGE_NAME }}:cache,mode=max
--- a/.github/workflows/python-publish.yaml
+++ b/.github/workflows/python-publish.yaml
@@ -1,4 +1,4 @@
-# This workflow uploads a Python Package to PyPI using Poetry when a release is created
+# This workflow will upload a Python Package using Twine when a release is created
 # For more information see: https://help.github.com/en/actions/language-and-framework-guides/using-python-with-github-actions#publishing-to-package-registries

 # This workflow uses actions that are not certified by GitHub.
@@ -11,6 +11,9 @@ name: Pypi
 on:
  release:
    types: [published]
+  push:
+    # branches: [ "main" ]
+    tags: [ "v*.*.*" ]

 permissions:
  contents: read
@@ -21,27 +24,30 @@ jobs:
    runs-on: ubuntu-latest

    steps:
-    - name: Checkout Repository
-      uses: actions/checkout@v4
+    - uses: actions/checkout@v3

-    - name: Set up Python
-      uses: actions/setup-python@v5
+    - name: Set up Python 3.10
+      uses: actions/setup-python@v4
      with:
-        python-version-file: pyproject.toml
-
-    - name: Install Poetry
-      run: |
-        pipx install "poetry>=2.0.0,<3.0.0"
+        python-version: "3.10"

    - name: Install dependencies
      run: |
-        poetry install --no-interaction --no-root
+        python -m pip install --upgrade --upgrade-strategy=eager pip setuptools wheel twine pipenv
+        python -m pip install -e . --upgrade
+        python -m pipenv install --dev --python 3.10
+      env:
+        PIPENV_DEFAULT_PYTHON_VERSION: "3.10"

-    - name: Build the package
+    - name: Build wheels
      run: |
-        poetry build
+        python -m pipenv run python setup.py sdist bdist_wheel

-    # Step 6: Publish to PyPI
-    - name: Publish to PyPI
-      run: |
-        poetry publish --username __token__ --password ${{ secrets.PYPI_API_TOKEN }}
+    - name: Publish a Python distribution to PyPI
+      uses: pypa/gh-action-pypi-publish@release/v1
+      with:
+        user: __token__
+        verbose: true
+        skip_existing: true
+        password: ${{ secrets.PYPI_API_TOKEN }}
+        packages_dir: dist/
--- a/.github/workflows/ruff.yaml
+++ b/.github/workflows/ruff.yaml
@@ -1,24 +0,0 @@
-name: Ruff Formatting & Linting
-
-on:
-  push:
-    branches: [ main ]
-  pull_request:
-    branches: [ main ]
-
-jobs:
-  build:
-    runs-on: ubuntu-latest
-    steps:
-      - uses: actions/checkout@v4
-      - name: Install Python
-        uses: actions/setup-python@v5
-        with:
-          python-version: "3.11"
-      - name: Install dependencies
-        run: |
-          python -m pip install --upgrade pip
-          pip install ruff
-
-      - name: Run Ruff
-        run: ruff check --output-format=github . && ruff format --check
--- a/.github/workflows/tests-core.yaml
+++ b/.github/workflows/tests-core.yaml
@@ -1,47 +0,0 @@
-name: Core Tests
-
-on:
-  push:
-    branches: [ main ]
-    paths:
-      - src/**
-      - poetry.lock
-      - pyproject.toml
-  pull_request:
-    paths:
-      - src/**
-      - poetry.lock
-      - pyproject.toml
-
-jobs:
-  tests:
-    runs-on: ${{ matrix.os }}
-    strategy:
-      fail-fast: false
-      matrix:
-        python-version: ["3.10", "3.11", "3.12"]
-        os: [ubuntu-22.04]
-        #TODO: re-enable ubuntu-latest, this is disabled as oscrypto cannot be pinned to github commit and pushed to pypi
-    defaults:
-      run:
-        working-directory: ./
-
-    steps:
-      - uses: actions/checkout@v4
-
-      - name: Install Poetry
-        run: pipx install poetry
-
-      - name: Set up Python ${{ matrix.python-version }}
-        uses: actions/setup-python@v5
-        with:
-          python-version: ${{ matrix.python-version }}
-          cache: 'poetry'
-
-      - name: Install dependencies
-        run: poetry install --no-interaction --with dev
-
-      - name: Run Core Tests
-        run: |
-          poetry run auto-archiver --version || true
-          poetry run pytest -ra -v -m "not download"
--- a/.github/workflows/tests-download.yaml
+++ b/.github/workflows/tests-download.yaml
@@ -1,40 +0,0 @@
-name: Download Tests
-
-on:
-  schedule:
-    - cron: '35 14 * * 1'
-  pull_request:
-    branches: [ main ]
-    paths:
-      - src/**
-
-jobs:
-  tests:
-    runs-on: ubuntu-22.04
-    strategy:
-      fail-fast: false
-      matrix:
-        python-version: ["3.10"] # only run expensive downloads on one (lowest) python version
-    defaults:
-      run:
-        working-directory: ./
-
-    steps:
-      - uses: actions/checkout@v4
-
-      - name: Install poetry
-        run: pipx install poetry
-
-      - name: Set up Python ${{ matrix.python-version }}
-        uses: actions/setup-python@v5
-        with:
-          python-version: ${{ matrix.python-version }}
-          cache: 'poetry'
-
-      - name: Install dependencies
-        run: poetry install --no-interaction --with dev
-
-      - name: Run Download Tests
-        run: poetry run pytest -ra -v -x -m "download"
-        env:
-          TWITTER_BEARER_TOKEN: ${{ secrets.TWITTER_BEARER_TOKEN }}
--- a/.gitignore
+++ b/.gitignore
@@ -27,12 +27,4 @@ instaloader.session
 orchestration.yaml
 auto_archiver.egg-info*
 logs*
-*.csv
-archived/
-dist*
-docs/_build/
-docs/source/autoapi/
-docs/source/modules/autogen/
-scripts/settings_page.html
-scripts/settings/src/schema.json 
-.vite
+*.csv
--- a/.pre-commit-config.yaml
+++ b/.pre-commit-config.yaml
@@ -1,7 +0,0 @@
-# Run Ruff formatter on commits.
-repos:
-  - repo: https://github.com/astral-sh/ruff-pre-commit
-    rev: v0.9.10
-    hooks:
-      - id: ruff
-      - id: ruff-format
--- a/.pylintrc
+++ b/.pylintrc
@@ -1,3 +0,0 @@
-[MAIN]
-
-ignore-patterns=(.*tests.*.py, __manifest__.py)
--- a/.readthedocs.yaml
+++ b/.readthedocs.yaml
@@ -1,28 +0,0 @@
-# Read the Docs configuration file
-# See https://docs.readthedocs.io/en/stable/config-file/v2.html for details
-
-# Required
-version: 2
-
-
-build:
-  os: ubuntu-22.04
-  tools:
-    python: "3.10"
-    nodejs: "22"
-  jobs:
-    post_install:
-      - pip install poetry
-      # https://python-poetry.org/docs/managing-dependencies/#dependency-groups
-      # VIRTUAL_ENV needs to be set manually for now.
-      # See https://github.com/readthedocs/readthedocs.org/pull/11152/
-      - VIRTUAL_ENV=$READTHEDOCS_VIRTUALENV_PATH poetry install --with docs
-
-      # generate the config editor page. Schema then HTML
-      - VIRTUAL_ENV=$READTHEDOCS_VIRTUALENV_PATH poetry run python scripts/generate_settings_schema.py
-      # install node dependencies and build the settings
-      - cd scripts/settings && npm install && npm run build && yes | cp -v dist/index.html ../../docs/source/installation/settings.html && cd ../..
-
-
-sphinx:
-  configuration: docs/source/conf.py
--- a/CONTRIBUTING.md
+++ b/CONTRIBUTING.md
@@ -1,49 +0,0 @@
-# Contributing to Auto Archiver
-
-Thank you for your interest in contributing to Auto Archiver! Your contributions help improve the project and make it more useful for everyone. Please follow the guidelines below to ensure a smooth collaboration.
-
-### 1. Reporting a Bug
-
-If you encounter a bug, please create an issue on GitHub with the following details:
-
-* Describe the bug: Provide a clear and concise description of the issue.
-* Steps to reproduce: Include the steps needed to reproduce the bug.
-* Expected behavior: Describe what you expected to happen.
-* Actual behavior: Explain what actually happened.
-* Screenshots/logs: If applicable, attach screenshots or logs to help diagnose the problem.
-* Environment: Mention the OS, Ruby version, and any other relevant details.
-
-### 2. Writing a Patch/Fix and Submitting Pull Requests
-
-If you’d like to fix a bug or improve existing code:
-
-1. Open a pull request on GitHub and link it to the relevant issue.
-2. Make sure to document your pull request with a clear description of what changes were made and why.
-3. Wait for review and make any requested changes.
-
-### 3. Creating New Modules
-
-If you want to add a new module to Auto Archiver:
-
-1. Ensure your module follows the existing [coding style and project structure](https://auto-archiver.readthedocs.io/en/latest/development/creating_modules.html).
-2. Write clear documentation explaining what your module does and how to use it.
-3. Ideally, include unit tests for your module!
-4. Follow the steps in Section 2 to submit a pull request.
-
-### 4. Do You Have Questions About the Source Code?
-
-If you have any questions about how the source code works or need help using Auto Archiver
-
-📝 Check the [Auto Archiver](https://auto-archiver.readthedocs.io/en/latest/) documentation.
-
-👉 Ask your questions in the [Bellingcat Discord](https://www.bellingcat.com/follow-bellingcat-on-social-media/).
-
-### 5. Do You Want to Contribute to the Documentation?
-
-We welcome contributions to the documentation!
-
-📖 Please read [Contributing to the Auto Archiver Documentation](https://auto-archiver.readthedocs.io/en/latest/development/docs.html) to learn how you can help improve the project's documentation.
-
------------------
-
-Thank you for contributing to Auto Archiver! 🚀
--- a/82
+++ b/82
@@ -1,71 +1,31 @@
-FROM webrecorder/browsertrix-crawler:1.4.2 AS base
+FROM webrecorder/browsertrix-crawler:latest

-ENV RUNNING_IN_DOCKER=1 \
-    LANG=C.UTF-8 \
-    LC_ALL=C.UTF-8 \
-    PYTHONDONTWRITEBYTECODE=1 \
-    PYTHONFAULTHANDLER=1 \
-    PATH="/root/.local/bin:$PATH"
-
-
-ARG TARGETARCH
-
-# Installing system dependencies
-RUN add-apt-repository ppa:mozillateam/ppa && \
-	apt-get update && \
-    apt-get install -y --no-install-recommends gcc ffmpeg fonts-noto exiftool && \
-	apt-get install -y --no-install-recommends firefox-esr && \
-    ln -s /usr/bin/firefox-esr /usr/bin/firefox
-
-ARG GECKODRIVER_VERSION=0.36.0
-
-RUN if [ $(uname -m) = "aarch64" ]; then \
-        GECKODRIVER_ARCH=linux-aarch64; \
-    else \
-        GECKODRIVER_ARCH=linux64; \
-    fi && \
-    wget https://github.com/mozilla/geckodriver/releases/download/v${GECKODRIVER_VERSION}/geckodriver-v${GECKODRIVER_VERSION}-${GECKODRIVER_ARCH}.tar.gz && \
-    tar -xvzf geckodriver* -C /usr/local/bin && \
-    chmod +x /usr/local/bin/geckodriver && \
-    rm geckodriver-v* && \
-    apt-get clean && \
-    rm -rf /var/lib/apt/lists/*
-
-
-# Poetry and runtime
-FROM base AS runtime
-
-ENV POETRY_NO_INTERACTION=1 \
-    POETRY_VIRTUALENVS_IN_PROJECT=1 \
-    POETRY_VIRTUALENVS_CREATE=1
-
-
-# Create a virtual environment for poetry and install it
-RUN python3 -m venv /poetry-venv && \
-    /poetry-venv/bin/python -m pip install --upgrade pip && \
-    /poetry-venv/bin/python -m pip install "poetry>=2.0.0,<3.0.0"
+ENV RUNNING_IN_DOCKER=1

 WORKDIR /app

-
-COPY pyproject.toml poetry.lock README.md ./
-# Copy dependency files and install dependencies (excluding the package itself)
-RUN /poetry-venv/bin/poetry install --only main --no-root --no-cache
+RUN pip install --upgrade pip && \
+	pip install pipenv && \
+	add-apt-repository ppa:mozillateam/ppa && \
+	apt-get update && \
+	apt-get install -y gcc ffmpeg fonts-noto exiftool && \
+	apt-get install -y --no-install-recommends firefox-esr && \
+	ln -s /usr/bin/firefox-esr /usr/bin/firefox && \
+	wget https://github.com/mozilla/geckodriver/releases/download/v0.33.0/geckodriver-v0.33.0-linux64.tar.gz && \
+	tar -xvzf geckodriver* -C /usr/local/bin && \
+	chmod +x /usr/local/bin/geckodriver && \
+	rm geckodriver-v*


-# Copy code: This is needed for poetry to install the package itself,
-# but the environment should be cached from the previous step if toml and lock files haven't changed
-COPY ./src/ .
-RUN /poetry-venv/bin/poetry install --only main --no-cache
+COPY Pipfile* ./
+# install from pipenv, with browsertrix-only requirements
+RUN pipenv install && \
+	pipenv install pywb uwsgi
+	
+# doing this at the end helps during development, builds are quick
+COPY ./src/ . 

-
-# Update PATH to include virtual environment binaries
-# Allowing entry point to run the application directly with Python
-ENV VIRTUAL_ENV=/app/.venv \
-    PATH="/app/.venv/bin:$PATH"
-
-ENTRYPOINT ["python3", "-m", "auto_archiver"]
+ENTRYPOINT ["pipenv", "run", "python3", "-m", "auto_archiver"]

 # should be executed with 2 volumes (3 if local_storage is used)
 # docker run --rm -v $PWD/secrets:/app/secrets -v $PWD/local_archive:/app/local_archive aa pipenv run python3 -m auto_archiver --config secrets/orchestration.yaml
-
--- a/79
+++ b/79
@@ -1,79 +0,0 @@
-# Variables
-SPHINXOPTS    ?=
-SPHINXBUILD   ?= sphinx-build
-SOURCEDIR     = docs/source
-BUILDDIR      = docs/_build
-
-.PHONY: help
-help:
-	@$(SPHINXBUILD) -M help "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(O)
-	@echo "Additional Commands:"
-	@echo "  make test         - Run all tests in 'tests/' with pytest"
-	@echo "  make ruff-check   - Run Ruff linting and formatting checks (safe)"
-	@echo "  make ruff-clean   - Auto-fix Ruff linting and formatting issues"
-	@echo "  make docs         - Generate documentation (same as 'make html')"
-	@echo "  make clean-docs   - Remove generated docs"
-	@echo "  make docker-build - Build the Auto Archiver Docker image"
-	@echo "  make docker-compose - Run Auto Archiver with Docker Compose"
-	@echo "  make docker-compose-rebuild - Rebuild and run Auto Archiver with Docker Compose"
-	@echo "  make show-docs    - Build and open the documentation in a browser"
-
-
-
-.PHONY: test
-test:
-	@echo "Running tests..."
-	@pytest tests --disable-warnings
-
-
-.PHONY: ruff-check
-ruff-check:
-	@echo "Checking code style with Ruff (safe)..."
-	@ruff check .
-
-
-.PHONY: ruff-clean
-ruff-clean:
-	@echo "Fixing lint and formatting issues with Ruff..."
-	@ruff check . --fix
-	@ruff format .
-
-
-.PHONY: docs
-docs:
-	@echo "Building documentation..."
-	@$(SPHINXBUILD) -M html "$(SOURCEDIR)" "$(BUILDDIR)"
-
-
-.PHONY: clean-docs
-clean-docs:
-	@echo "Cleaning up generated documentation files..."
-	@$(SPHINXBUILD) -M clean "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(O)
-	@rm -rf "$(SOURCEDIR)/autoapi/" "$(SOURCEDIR)/modules/autogen/"
-	@echo "Cleanup complete."
-
-
-.PHONY: show-docs
-show-docs:
-	@echo "Opening documentation in browser..."
-	@open "$(BUILDDIR)/html/index.html"
-
-.PHONY: docker-build
-docker-build:
-	@echo "Building local Auto Archiver Docker image..."
-	@docker compose build  # Uses the same build context as docker-compose.yml
-
-.PHONY: docker-compose
-docker-compose:
-	@echo "Running Auto Archiver with Docker Compose..."
-	@docker compose up
-
-.PHONY: docker-compose-rebuild
-docker-compose-rebuild:
-	@echo "Rebuilding and running Auto Archiver with Docker Compose..."
-	@docker compose up --build
-
-# Catch-all for Sphinx commands
-.PHONY: Makefile
-%: Makefile
-	@$(SPHINXBUILD) -M $@ "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(O)
--- a/45
+++ b/45
@@ -0,0 +1,45 @@
+[[source]]
+url = "https://pypi.org/simple"
+verify_ssl = true
+name = "pypi"
+
+[packages]
+gspread = "*"
+boto3 = "*"
+argparse = "*"
+beautifulsoup4 = "*"
+tiktok-downloader = "*"
+bs4 = "*"
+loguru = "*"
+ffmpeg-python = "*"
+selenium = "*"
+snscrape = "*"
+telethon = "*"
+google-api-python-client = "*"
+google-auth-httplib2 = "*"
+google-auth-oauthlib = "*"
+oauth2client = "*"
+pdqhash = "*"
+pillow = "*"
+python-slugify = "*"
+pyyaml = "*"
+dateparser = "*"
+python-twitter-v2 = "*"
+instaloader = "*"
+tqdm = "*"
+jinja2 = "*"
+cryptography = "*"
+dataclasses-json = "*"
+yt-dlp = "*"
+vk-url-scraper = "*"
+requests = {extras = ["socks"], version = "*"}
+numpy = "*"
+warcio = "*"
+jsonlines = "*"
+
+[dev-packages]
+autopep8 = "*"
+setuptools-pipfile = "*"
+
+[requires]
+python_version = "3.10"
--- a/Pipfile.lock
+++ b/Pipfile.lock
--- a/README.md
+++ b/README.md
@@ -2,40 +2,243 @@

 [![PyPI version](https://badge.fury.io/py/auto-archiver.svg)](https://badge.fury.io/py/auto-archiver)
 [![Docker Image Version (latest by date)](https://img.shields.io/docker/v/bellingcat/auto-archiver?label=version&logo=docker)](https://hub.docker.com/r/bellingcat/auto-archiver)
-[![Core Test Status](https://github.com/bellingcat/auto-archiver/workflows/Core%20Tests/badge.svg)](https://github.com/bellingcat/auto-archiver/actions/workflows/tests-core.yaml)
-[![Download Test Status](https://github.com/bellingcat/auto-archiver/workflows/Download%20Tests/badge.svg)](https://github.com/bellingcat/auto-archiver/actions/workflows/tests-download.yaml)
 <!-- ![Docker Pulls](https://img.shields.io/docker/pulls/bellingcat/auto-archiver) -->
 <!-- [![PyPI download month](https://img.shields.io/pypi/dm/auto-archiver.svg)](https://pypi.python.org/pypi/auto-archiver/) -->
 <!-- [![Documentation Status](https://readthedocs.org/projects/vk-url-scraper/badge/?version=latest)](https://vk-url-scraper.readthedocs.io/en/latest/?badge=latest) -->


-
-Auto Archiver is a Python tool to automatically archive content on the web in a secure and verifiable way. It takes URLs from different sources (e.g. a CSV file, Google Sheets, command line etc.) and archives the content of each one. It can archive social media posts, videos, images and webpages. Content can be enriched, then saved either locally or remotely (S3 bucket, Google Drive). The status of the archiving process can be appended to a CSV report, or if using Google Sheets – back to the original sheet.
-
-<div class="hidden_rtd">
-  
-**[See the Auto Archiver documentation for more information.](https://auto-archiver.readthedocs.io/en/latest/)**
-
-</div>
-
 Read the [article about Auto Archiver on bellingcat.com](https://www.bellingcat.com/resources/2022/09/22/preserve-vital-online-content-with-bellingcats-auto-archiver-tool/).


-## Installation
+Python tool to automatically archive social media posts, videos, and images from a Google Sheets, the console, and more. Uses different archivers depending on the platform, and can save content to local storage, S3 bucket (Digital Ocean Spaces, AWS, ...), and Google Drive. If using Google Sheets as the source for links, it will be updated with information about the archived content. It can be run manually or on an automated basis.

-View the [Installation Guide](https://auto-archiver.readthedocs.io/en/latest/installation/installation.html) for full instructions
+There are 3 ways to use the auto-archiver:
+1. (easiest installation) via docker
+2. (local python install) `pip install auto-archiver`
+3. (legacy/development) clone and manually install from repo (see legacy [tutorial video](https://youtu.be/VfAhcuV2tLQ))

-**Advanced:**
+But **you always need a configuration/orchestration file**, which is where you'll configure where/what/how to archive. Make sure you read [orchestration](#orchestration).

-To get started quickly using Docker:

-`docker pull bellingcat/auto-archiver && docker run -it --rm -v secrets:/app/secrets bellingcat/auto-archiver --config secrets/orchestration.yaml`
+## How to install and run the auto-archiver

-Or pip:
+### Option 1 - docker

-`pip install auto-archiver && auto-archiver --help`
+[![dockeri.co](https://dockerico.blankenship.io/image/bellingcat/auto-archiver)](https://hub.docker.com/r/bellingcat/auto-archiver)

-## Contributing
+Docker works like a virtual machine running inside your computer, it isolates everything and makes installation simple. Since it is an isolated environment when you need to pass it your orchestration file or get downloaded media out of docker you will need to connect folders on your machine with folders inside docker with the `-v` volume flag.

-We welcome contributions to the Auto Archiver project! See the [Contributing Guide](https://auto-archiver.readthedocs.io/en/latest/contributing.html) for how to get involved!

+1. install [docker](https://docs.docker.com/get-docker/)
+2. pull the auto-archiver docker [image](https://hub.docker.com/r/bellingcat/auto-archiver) with `docker pull bellingcat/auto-archiver`
+3. run the docker image locally in a container: `docker run --rm -v $PWD/secrets:/app/secrets -v $PWD/local_archive:/app/local_archive bellingcat/auto-archiver --config secrets/orchestration.yaml` breaking this command down:
+   1. `docker run` tells docker to start a new container (an instance of the image)
+   2. `--rm` makes sure this container is removed after execution (less garbage locally)
+   3. `-v $PWD/secrets:/app/secrets` - your secrets folder
+      1. `-v` is a volume flag which means a folder that you have on your computer will be connected to a folder inside the docker container
+      2. `$PWD/secrets` points to a `secrets/` folder in your current working directory (where your console points to), we use this folder as a best practice to hold all the secrets/tokens/passwords/... you use
+      3. `/app/secrets` points to the path the docker container where this image can be found
+   4.  `-v $PWD/local_archive:/app/local_archive` - (optional) if you use local_storage
+       1.  `-v` same as above, this is a volume instruction
+       2.  `$PWD/local_archive` is a folder `local_archive/` in case you want to archive locally and have the files accessible outside docker
+       3.  `/app/local_archive` is a folder inside docker that you can reference in your orchestration.yml file 
+
+### Option 2 - python package
+
+<details><summary><code>Python package instructions</code></summary>
+
+1. make sure you have python 3.8 or higher installed
+2. install the package `pip/pipenv/conda install auto-archiver`
+3. test it's installed with `auto-archiver --help`
+4. run it with your orchestration file and pass any flags you want in the command line `auto-archiver --config secrets/orchestration.yaml` if your orchestration file is inside a `secrets/`, which we advise
+   
+You will also need [ffmpeg](https://www.ffmpeg.org/), [firefox](https://www.mozilla.org/en-US/firefox/new/) and [geckodriver](https://github.com/mozilla/geckodriver/releases), and optionally [fonts-noto](https://fonts.google.com/noto). Similar to the local installation. 
+
+</details>
+
+
+### Option 3 - local installation
+This can also be used for development.
+
+<details><summary><code>Legacy instructions, only use if docker/package is not an option</code></summary>
+
+
+Install the following locally:
+1. [ffmpeg](https://www.ffmpeg.org/) must also be installed locally for this tool to work. 
+2. [firefox](https://www.mozilla.org/en-US/firefox/new/) and [geckodriver](https://github.com/mozilla/geckodriver/releases) on a path folder like `/usr/local/bin`. 
+3. (optional) [fonts-noto](https://fonts.google.com/noto) to deal with multiple unicode characters during selenium/geckodriver's screenshots: `sudo apt install fonts-noto -y`. 
+
+Clone and run:
+1. `git clone https://github.com/bellingcat/auto-archiver`
+2. `pipenv install`
+3. `pipenv run python -m src.auto_archiver --config secrets/orchestration.yaml`
+
+
+</details><br/>
+
+# Orchestration
+The archiver work is orchestrated by the following workflow (we call each a **step**): 
+1. **Feeder** gets the links (from a spreadsheet, from the console, ...)
+2. **Archiver** tries to archive the link (twitter, youtube, ...)
+3. **Enricher** adds more info to the content (hashes, thumbnails, ...)
+4. **Formatter** creates a report from all the archived content (HTML, PDF, ...)
+5. **Database** knows what's been archived and also stores the archive result (spreadsheet, CSV, or just the console)
+
+To setup an auto-archiver instance create an `orchestration.yaml` which contains the workflow you would like. We advise you put this file into a `secrets/` folder and do not share it with others because it will contain passwords and other secrets. 
+
+The structure of orchestration file is split into 2 parts: `steps` (what **steps** to use) and `configurations` (how those steps should behave), here's a simplification:
+```yaml
+# orchestration.yaml content
+steps:
+  feeder: gsheet_feeder
+  archivers: # order matters
+    - youtubedl_archiver
+  enrichers:
+    - thumbnail_enricher
+  formatter: html_formatter
+  storages:
+    - local_storage
+  databases:
+    - gsheet_db
+
+configurations:
+  gsheet_feeder:
+    sheet: "your google sheet name"
+    header: 2 # row with header for your sheet
+  # ... configurations for the other steps here ...
+```
+
+To see all available `steps` (which archivers, storages, databses, ...) exist check the [example.orchestration.yaml](example.orchestration.yaml).
+
+All the `configurations` in the `orchestration.yaml` file (you can name it differently but need to pass it in the `--config FILENAME` argument) can be seen in the console by using the `--help` flag. They can also be overwritten, for example if you are using the `cli_feeder` to archive from the command line and want to provide the URLs you should do:
+
+```bash
+auto-archiver --config secrets/orchestration.yaml --cli_feeder.urls="url1,url2,url3"
+```
+
+Here's the complete workflow that the auto-archiver goes through:
+```mermaid
+graph TD
+    s((start)) --> F(fa:fa-table Feeder)
+    F -->|get and clean URL| D1{fa:fa-database Database}
+    D1 -->|is already archived| e((end))
+    D1 -->|not yet archived| a(fa:fa-download Archivers)
+    a -->|got media| E(fa:fa-chart-line Enrichers)
+    E --> S[fa:fa-box-archive Storages]
+    E --> Fo(fa:fa-code Formatter)
+    Fo --> S
+    Fo -->|update database| D2(fa:fa-database Database)
+    D2 --> e
+```
+
+## Orchestration checklist
+Use this to make sure you help making sure you did all the required steps:
+* [ ] you have a `/secrets` folder with all your configuration files including
+  * [ ] a orchestration file eg: `orchestration.yaml` pointing to the correct location of other files
+  * [ ] (optional if you use GoogleSheets) you have a `service_account.json` (see [how-to](https://gspread.readthedocs.io/en/latest/oauth2.html#for-bots-using-service-account))
+  * [ ] (optional for telegram) a `anon.session` which appears after the 1st run where you login to telegram
+    * if you use private channels you need to add `channel_invites` and set `join_channels=true` at least once
+  * [ ] (optional for VK) a `vk_config.v2.json`
+  * [ ] (optional for using GoogleDrive storage) `gd-token.json` (see [help script](scripts/create_update_gdrive_oauth_token.py))
+  * [ ] (optional for instagram) `instaloader.session` file which appears after the 1st run and login in instagram
+  * [ ] (optional for browsertrix) `profile.tar.gz` file
+
+#### Example invocations
+The recommended way to run the auto-archiver is through Docker. The invocations below will run the auto-archiver Docker image using a configuration file that you have specified
+
+```bash
+# all the configurations come from ./secrets/orchestration.yaml
+docker run --rm -v $PWD/secrets:/app/secrets -v $PWD/local_archive:/app/local_archive bellingcat/auto-archiver --config secrets/orchestration.yaml
+# uses the same configurations but for another google docs sheet 
+# with a header on row 2 and with some different column names
+# notice that columns is a dictionary so you need to pass it as JSON and it will override only the values provided
+docker run --rm -v $PWD/secrets:/app/secrets -v $PWD/local_archive:/app/local_archive bellingcat/auto-archiver --config secrets/orchestration.yaml --gsheet_feeder.sheet="use it on another sheets doc" --gsheet_feeder.header=2 --gsheet_feeder.columns='{"url": "link"}'
+# all the configurations come from orchestration.yaml and specifies that s3 files should be private
+docker run --rm -v $PWD/secrets:/app/secrets -v $PWD/local_archive:/app/local_archive bellingcat/auto-archiver --config secrets/orchestration.yaml --s3_storage.private=1
+```
+
+The auto-archiver can also be run locally, if pre-requisites are correctly configured. Equivalent invocations are below.
+
+```bash
+# all the configurations come from ./secrets/orchestration.yaml
+auto-archiver --config secrets/orchestration.yaml
+# uses the same configurations but for another google docs sheet 
+# with a header on row 2 and with some different column names
+# notice that columns is a dictionary so you need to pass it as JSON and it will override only the values provided
+auto-archiver --config secrets/orchestration.yaml --gsheet_feeder.sheet="use it on another sheets doc" --gsheet_feeder.header=2 --gsheet_feeder.columns='{"url": "link"}'
+# all the configurations come from orchestration.yaml and specifies that s3 files should be private
+auto-archiver --config secrets/orchestration.yaml --s3_storage.private=1
+```
+
+### Extra notes on configuration
+#### Google Drive
+To use Google Drive storage you need the id of the shared folder in the `config.yaml` file which must be shared with the service account eg `autoarchiverservice@auto-archiver-111111.iam.gserviceaccount.com` and then you can use `--storage=gd`
+
+#### Telethon + Instagram with telegram bot
+The first time you run, you will be prompted to do a authentication with the phone number associated, alternatively you can put your `anon.session` in the root.
+
+
+## Running on Google Sheets Feeder (gsheet_feeder)
+The `--gseets_feeder.sheet` property is the name of the Google Sheet to check for URLs. 
+This sheet must have been shared with the Google Service account used by `gspread`. 
+This sheet must also have specific columns (case-insensitive) in the `header` as specified in [Gsheet.configs](src/auto_archiver/utils/gsheet.py). The default names of these columns and their purpose is:
+
+Inputs:
+
+* **Link** *(required)*: the URL of the post to archive
+* **Destination folder**: custom folder for archived file (regardless of storage)
+
+Outputs:
+* **Archive status** *(required)*: Status of archive operation
+* **Archive location**: URL of archived post
+* **Archive date**: Date archived
+* **Thumbnail**: Embeds a thumbnail for the post in the spreadsheet
+* **Timestamp**: Timestamp of original post
+* **Title**: Post title
+* **Text**: Post text
+* **Screenshot**: Link to screenshot of post
+* **Hash**: Hash of archived HTML file (which contains hashes of post media) - for checksums/verification
+* **Perceptual Hash**: Perceptual hashes of found images - these can be used for de-duplication of content
+* **WACZ**: Link to a WACZ web archive of post
+* **ReplayWebpage**: Link to a ReplayWebpage viewer of the WACZ archive
+
+For example, this is a spreadsheet configured with all of the columns for the auto archiver and a few URLs to archive. (Note that the column names are not case sensitive.)
+
+![A screenshot of a Google Spreadsheet with column headers defined as above, and several Youtube and Twitter URLs in the "Link" column](docs/demo-before.png)
+
+Now the auto archiver can be invoked, with this command in this example: `docker run --rm -v $PWD/secrets:/app/secrets -v $PWD/local_archive:/app/local_archive bellingcat/auto-archiver:dockerize --config secrets/orchestration-global.yaml --gsheet_feeder.sheet "Auto archive test 2023-2"`. Note that the sheet name has been overridden/specified in the command line invocation.
+
+When the auto archiver starts running, it updates the "Archive status" column.
+
+![A screenshot of a Google Spreadsheet with column headers defined as above, and several Youtube and Twitter URLs in the "Link" column. The auto archiver has added "archive in progress" to one of the status columns.](docs/demo-progress.png)
+
+The links are downloaded and archived, and the spreadsheet is updated to the following:
+
+![A screenshot of a Google Spreadsheet with videos archived and metadata added per the description of the columns above.](docs/demo-after.png)
+
+Note that the first row is skipped, as it is assumed to be a header row (`--gsheet_feeder.header=1` and you can change it if you use more rows above). Rows with an empty URL column, or a non-empty archive column are also skipped. All sheets in the document will be checked.
+
+The "archive location" link contains the path of the archived file, in local storage, S3, or in Google Drive.
+
+![The archive result for a link in the demo sheet.](docs/demo-archive.png)
+
+---
+## Development
+Use `python -m src.auto_archiver --config secrets/orchestration.yaml` to run from the local development environment.
+
+#### Docker development
+working with docker locally:
+  * `docker build . -t auto-archiver` to build a local image
+  * `docker run --rm -v $PWD/secrets:/app/secrets auto-archiver  --config secrets/orchestration.yaml`
+    * to use local archive, also create a volume `-v` for it by adding `-v $PWD/local_archive:/app/local_archive`
+
+
+release to docker hub
+  * `docker image tag auto-archiver bellingcat/auto-archiver:latest`
+  * `docker push bellingcat/auto-archiver`
+
+#### RELEASE
+* update version in [version.py](src/auto_archiver/version.py)
+* run `bash ./scripts/release.sh` and confirm
+* package is automatically updated in pypi
+* docker image is automatically pushed to dockerhup
--- a/docker-compose.yaml
+++ b/docker-compose.yaml
@@ -1,16 +0,0 @@
-version: '3.8'
-
-services:
-  auto-archiver:
-  # point to the local dockerfile
-    build:
-      context: .
-      dockerfile: Dockerfile
-    container_name: auto-archiver
-    volumes:
-      - ./secrets:/app/secrets
-      - ./local_archive:/app/local_archive
-    environment:
-      - WACZ_ENABLE_DOCKER=true
-      - RUNNING_IN_DOCKER=true
-    command: --config secrets/orchestration.yaml
--- a/docs/_static/custom.css
+++ b/docs/_static/custom.css
@@ -1,4 +0,0 @@
-.hidden_rtd {
-    display:none;
-}
-
--- a/docs/_templates/autoapi/index.rst
+++ b/docs/_templates/autoapi/index.rst
@@ -1,46 +0,0 @@
-API Reference
-=============
-
-These pages are intended for developers of the `auto-archiver` package, 
-and include documentation on the core classes and functions used by 
-the auto-archiver
-
-
-Core Classes
------------
-
-
-.. toctree::
-   :titlesonly:
-
-   {% for page in pages|selectattr("is_top_level_object") %}
-   {% if page.name == 'core' %}
-   {{ page.include_path }}
-   {% endif %}
-   {% endfor %}
-
-Util Functions
--------------
-
-.. toctree::
-   :titlesonly:
-
-   {% for page in pages|selectattr("is_top_level_object") %}
-   {% if page.name == 'utils' %}
-   {{ page.include_path }}
-   {% endif %}
-   {% endfor %}
-
-
-Core Modules
------------
-
-.. toctree::
-   :titlesonly:
-
-   {% for page in pages|selectattr("is_top_level_object") %}
-   {% if page.name != 'core' and page.name != 'utils' %}
-   {{ page.include_path }}
-   {% endif %}
-   {% endfor %}
-
--- a/docs/_templates/autoapi/python/attribute.rst
+++ b/docs/_templates/autoapi/python/attribute.rst
@@ -1 +0,0 @@
-{% extends "python/data.rst" %}
--- a/docs/_templates/autoapi/python/class.rst
+++ b/docs/_templates/autoapi/python/class.rst
@@ -1,104 +0,0 @@
-{% if obj.display %}
-   {% if is_own_page %}
-{{ obj.id }}
-{{ "=" * obj.id | length }}
-
-   {% endif %}
-   {% set visible_children = obj.children|selectattr("display")|list %}
-   {% set own_page_children = visible_children|selectattr("type", "in", own_page_types)|list %}
-   {% if is_own_page and own_page_children %}
-.. toctree::
-   :hidden:
-
-      {% for child in own_page_children %}
-   {{ child.include_path }}
-      {% endfor %}
-
-   {% endif %}
-.. py:{{ obj.type }}:: {% if is_own_page %}{{ obj.id }}{% else %}{{ obj.short_name }}{% endif %}{% if obj.args %}({{ obj.args }}){% endif %}
-
-   {% for (args, return_annotation) in obj.overloads %}
-      {{ " " * (obj.type | length) }}   {{ obj.short_name }}{% if args %}({{ args }}){% endif %}
-
-   {% endfor %}
-   {% if obj.bases %}
-      {% if "show-inheritance" in autoapi_options %}
-
-   Bases: {% for base in obj.bases %}{{ base|link_objs }}{% if not loop.last %}, {% endif %}{% endfor %}
-      {% endif %}
-
-
-      {% if "show-inheritance-diagram" in autoapi_options and obj.bases != ["object"] %}
-   .. autoapi-inheritance-diagram:: {{ obj.obj["full_name"] }}
-      :parts: 1
-         {% if "private-members" in autoapi_options %}
-      :private-bases:
-         {% endif %}
-
-      {% endif %}
-   {% endif %}
-   {% if obj.docstring %}
-
-   {{ obj.docstring|indent(3) }}
-   {% endif %}
-   {% for obj_item in visible_children %}
-      {% if obj_item.type not in own_page_types %}
-
-   {{ obj_item.render()|indent(3) }}
-      {% endif %}
-   {% endfor %}
-   {% if is_own_page and own_page_children %}
-      {% set visible_attributes = own_page_children|selectattr("type", "equalto", "attribute")|list %}
-      {% if visible_attributes %}
-Attributes
----------
-
-.. autoapisummary::
-
-         {% for attribute in visible_attributes %}
-   {{ attribute.id }}
-         {% endfor %}
-
-
-      {% endif %}
-      {% set visible_exceptions = own_page_children|selectattr("type", "equalto", "exception")|list %}
-      {% if visible_exceptions %}
-Exceptions
----------
-
-.. autoapisummary::
-
-         {% for exception in visible_exceptions %}
-   {{ exception.id }}
-         {% endfor %}
-
-
-      {% endif %}
-      {% set visible_classes = own_page_children|selectattr("type", "equalto", "class")|list %}
-      {% if visible_classes %}
-Classes
-------
-
-.. autoapisummary::
-
-         {% for klass in visible_classes %}
-   {{ klass.id }}
-         {% endfor %}
-
-
-      {% endif %}
-      {% set visible_methods = own_page_children|selectattr("type", "equalto", "method")|list %}
-      {% if visible_methods %}
-Methods
-------
-
-.. autoapisummary::
-
-            {% for method in visible_methods %}
-   {{ method.id }}
-            {% endfor %}
-
-
-      {% endif %}
-   {% endif %}
-{% endif %}
--- a/docs/_templates/autoapi/python/data.rst
+++ b/docs/_templates/autoapi/python/data.rst
@@ -1,38 +0,0 @@
-{% if obj.display %}
-   {% if is_own_page %}
-{{ obj.id }}
-{{ "=" * obj.id | length }}
-
-   {% endif %}
-.. py:{{ obj.type }}:: {% if is_own_page %}{{ obj.id }}{% else %}{{ obj.name }}{% endif %}
-   {% if obj.annotation is not none %}
-
-   :type: {% if obj.annotation %} {{ obj.annotation }}{% endif %}
-   {% endif %}
-   {% if obj.value is not none %}
-
-      {% if obj.value.splitlines()|count > 1 %}
-   :value: Multiline-String
-
-   .. raw:: html
-
-      <details><summary>Show Value</summary>
-
-   .. code-block:: python
-
-      {{ obj.value|indent(width=6,blank=true) }}
-
-   .. raw:: html
-
-      </details>
-
-      {% else %}
-   :value: {{ obj.value|truncate(100) }}
-      {% endif %}
-   {% endif %}
-
-   {% if obj.docstring %}
-
-   {{ obj.docstring|indent(3) }}
-   {% endif %}
-{% endif %}
--- a/docs/_templates/autoapi/python/exception.rst
+++ b/docs/_templates/autoapi/python/exception.rst
@@ -1 +0,0 @@
-{% extends "python/class.rst" %}
--- a/docs/_templates/autoapi/python/function.rst
+++ b/docs/_templates/autoapi/python/function.rst
@@ -1,21 +0,0 @@
-{% if obj.display %}
-   {% if is_own_page %}
-{{ obj.id }}
-{{ "=" * obj.id | length }}
-
-   {% endif %}
-.. py:function:: {% if is_own_page %}{{ obj.id }}{% else %}{{ obj.short_name }}{% endif %}({{ obj.args }}){% if obj.return_annotation is not none %} -> {{ obj.return_annotation }}{% endif %}
-   {% for (args, return_annotation) in obj.overloads %}
-
-                 {%+ if is_own_page %}{{ obj.id }}{% else %}{{ obj.short_name }}{% endif %}({{ args }}){% if return_annotation is not none %} -> {{ return_annotation }}{% endif %}
-   {% endfor %}
-   {% for property in obj.properties %}
-
-   :{{ property }}:
-   {% endfor %}
-
-   {% if obj.docstring %}
-
-   {{ obj.docstring|indent(3) }}
-   {% endif %}
-{% endif %}
--- a/docs/_templates/autoapi/python/method.rst
+++ b/docs/_templates/autoapi/python/method.rst
@@ -1,21 +0,0 @@
-{% if obj.display %}
-   {% if is_own_page %}
-{{ obj.id }}
-{{ "=" * obj.id | length }}
-
-   {% endif %}
-.. py:method:: {% if is_own_page %}{{ obj.id }}{% else %}{{ obj.short_name }}{% endif %}({{ obj.args }}){% if obj.return_annotation is not none %} -> {{ obj.return_annotation }}{% endif %}
-   {% for (args, return_annotation) in obj.overloads %}
-
-               {%+ if is_own_page %}{{ obj.id }}{% else %}{{ obj.short_name }}{% endif %}({{ args }}){% if return_annotation is not none %} -> {{ return_annotation }}{% endif %}
-   {% endfor %}
-   {% for property in obj.properties %}
-
-   :{{ property }}:
-   {% endfor %}
-
-   {% if obj.docstring %}
-
-   {{ obj.docstring|indent(3) }}
-   {% endif %}
-{% endif %}
--- a/docs/_templates/autoapi/python/module.rst
+++ b/docs/_templates/autoapi/python/module.rst
@@ -1,156 +0,0 @@
-{% if obj.display %}
-   {% if is_own_page %}
-{{ obj.id }}
-{{ "=" * obj.id|length }}
-
-.. py:module:: {{ obj.name }}
-
-      {% if obj.docstring %}
-.. autoapi-nested-parse::
-
-   {{ obj.docstring|indent(3) }}
-
-      {% endif %}
-
-      {% block submodules %}
-         {% set visible_subpackages = obj.subpackages|selectattr("display")|list %}
-         {% set visible_submodules = obj.submodules|selectattr("display")|list %}
-         {% set visible_submodules = (visible_subpackages + visible_submodules)|sort %}
-         {% if visible_submodules %}
-Submodules
----------
-
-.. toctree::
-   :maxdepth: 1
-
-            {% for submodule in visible_submodules %}
-   {{ submodule.include_path }}
-            {% endfor %}
-
-
-         {% endif %}
-      {% endblock %}
-      {% block content %}
-         {% set visible_children = obj.children|selectattr("display")|list %}
-         {% if visible_children %}
-            {% set visible_attributes = visible_children|selectattr("type", "equalto", "data")|list %}
-            {% if visible_attributes %}
-               {% if "attribute" in own_page_types or "show-module-summary" in autoapi_options %}
-Attributes
----------
-
-                  {% if "attribute" in own_page_types %}
-.. toctree::
-   :hidden:
-
-                     {% for attribute in visible_attributes %}
-   {{ attribute.include_path }}
-                     {% endfor %}
-
-                  {% endif %}
-.. autoapisummary::
-
-                  {% for attribute in visible_attributes %}
-   {{ attribute.id }}
-                  {% endfor %}
-               {% endif %}
-
-
-            {% endif %}
-            {% set visible_exceptions = visible_children|selectattr("type", "equalto", "exception")|list %}
-            {% if visible_exceptions %}
-               {% if "exception" in own_page_types or "show-module-summary" in autoapi_options %}
-Exceptions
----------
-
-                  {% if "exception" in own_page_types %}
-.. toctree::
-   :hidden:
-
-                     {% for exception in visible_exceptions %}
-   {{ exception.include_path }}
-                     {% endfor %}
-
-                  {% endif %}
-.. autoapisummary::
-
-                  {% for exception in visible_exceptions %}
-   {{ exception.id }}
-                  {% endfor %}
-               {% endif %}
-
-
-            {% endif %}
-            {% set visible_classes = visible_children|selectattr("type", "equalto", "class")|list %}
-            {% if visible_classes %}
-               {% if "class" in own_page_types or "show-module-summary" in autoapi_options %}
-Classes
-------
-
-                  {% if "class" in own_page_types %}
-.. toctree::
-   :hidden:
-
-                     {% for klass in visible_classes %}
-   {{ klass.include_path }}
-                     {% endfor %}
-
-                  {% endif %}
-.. autoapisummary::
-
-                  {% for klass in visible_classes %}
-   {{ klass.id }}
-                  {% endfor %}
-               {% endif %}
-
-
-            {% endif %}
-            {% set visible_functions = visible_children|selectattr("type", "equalto", "function")|list %}
-            {% if visible_functions %}
-               {% if "function" in own_page_types or "show-module-summary" in autoapi_options %}
-Functions
---------
-
-                  {% if "function" in own_page_types %}
-.. toctree::
-   :hidden:
-
-                     {% for function in visible_functions %}
-   {{ function.include_path }}
-                     {% endfor %}
-
-                  {% endif %}
-.. autoapisummary::
-
-                  {% for function in visible_functions %}
-   {{ function.id }}
-                  {% endfor %}
-               {% endif %}
-
-
-            {% endif %}
-            {% set this_page_children = visible_children|rejectattr("type", "in", own_page_types)|list %}
-            {% if this_page_children %}
-{{ obj.type|title }} Contents
-{{ "-" * obj.type|length }}---------
-
-               {% for obj_item in this_page_children %}
-{{ obj_item.render()|indent(0) }}
-               {% endfor %}
-            {% endif %}
-         {% endif %}
-      {% endblock %}
-   {% else %}
-.. py:module:: {{ obj.name }}
-
-      {% if obj.docstring %}
-   .. autoapi-nested-parse::
-
-      {{ obj.docstring|indent(6) }}
-
-      {% endif %}
-      {% for obj_item in visible_children %}
-   {{ obj_item.render()|indent(3) }}
-      {% endfor %}
-   {% endif %}
-{% endif %}
--- a/docs/_templates/autoapi/python/package.rst
+++ b/docs/_templates/autoapi/python/package.rst
@@ -1 +0,0 @@
-{% extends "python/module.rst" %}
--- a/docs/_templates/autoapi/python/property.rst
+++ b/docs/_templates/autoapi/python/property.rst
@@ -1,21 +0,0 @@
-{% if obj.display %}
-   {% if is_own_page %}
-{{ obj.id }}
-{{ "=" * obj.id | length }}
-
-   {% endif %}
-.. py:property:: {% if is_own_page %}{{ obj.id}}{% else %}{{ obj.short_name }}{% endif %}
-   {% if obj.annotation %}
-
-   :type: {{ obj.annotation }}
-   {% endif %}
-   {% for property in obj.properties %}
-
-   :{{ property }}:
-   {% endfor %}
-
-   {% if obj.docstring %}
-
-   {{ obj.docstring|indent(3) }}
-   {% endif %}
-{% endif %}
--- a/docs/scripts/init.py
+++ b/docs/scripts/init.py
@@ -1 +0,0 @@
-from scripts import generate_module_docs
--- a/docs/scripts/scripts.py
+++ b/docs/scripts/scripts.py
@@ -1,148 +0,0 @@
-# iterate through all the modules in auto_archiver.modules and turn the __manifest__.py file into a markdown table
-from pathlib import Path
-from auto_archiver.core.module import ModuleFactory
-from auto_archiver.core.base_module import BaseModule
-from ruamel.yaml import YAML
-from ruamel.yaml.comments import CommentedMap
-import io
-
-MODULES_FOLDER = Path(__file__).parent.parent.parent.parent / "src" / "auto_archiver" / "modules"
-SAVE_FOLDER = Path(__file__).parent.parent / "source" / "modules" / "autogen"
-
-type_color = {
-    "feeder": "<span style='color: #FFA500'>[feeder](/core_modules.md#feeder-modules)</a></span>",
-    "extractor": "<span style='color: #00FF00'>[extractor](/core_modules.md#extractor-modules)</a></span>",
-    "enricher": "<span style='color: #0000FF'>[enricher](/core_modules.md#enricher-modules)</a></span>",
-    "database": "<span style='color: #FF00FF'>[database](/core_modules.md#database-modules)</a></span>",
-    "storage": "<span style='color: #FFFF00'>[storage](/core_modules.md#storage-modules)</a></span>",
-    "formatter": "<span style='color: #00FFFF'>[formatter](/core_modules.md#formatter-modules)</a></span>",
-}
-
-TABLE_HEADER = ("Option", "Description", "Default", "Type")
-
-EXAMPLE_YAML = """
-# steps configuration
-steps:
-...
-{steps_str}
-...
-
-# module configuration
-...
-
-{config_string}
-
-"""
-
-
-def generate_module_docs():
-    yaml = YAML()
-    SAVE_FOLDER.mkdir(exist_ok=True)
-    modules_by_type = {}
-
-    header_row = "| " + " | ".join(TABLE_HEADER) + "|\n" + "| --- " * len(TABLE_HEADER) + "|\n"
-    global_table = "\n## Configuration Options\n" + header_row
-
-    global_yaml = yaml.load("""\n# Module configuration\nplaceholder: {}""")
-
-    for module in sorted(ModuleFactory().available_modules(), key=lambda x: (x.requires_setup, x.name)):
-        # generate the markdown file from the __manifest__.py file.
-
-        manifest = module.manifest
-        for type in manifest["type"]:
-            modules_by_type.setdefault(type, []).append(module)
-
-        description = "\n".join(line.lstrip() for line in manifest["description"].split("\n"))
-        types = ", ".join(type_color[t] for t in manifest["type"])
-        readme_str = f"""
-# {manifest["name"]}
-```{{admonition}} Module type
-
-{types}
-```
-{description}
-"""
-        steps_str = "\n".join(f"  {t}s:\n  - {module.name}" for t in manifest["type"])
-
-        if not manifest["configs"]:
-            config_string = f"# No configuration options for {module.name}.*\n"
-        else:
-            config_table = header_row
-            config_yaml = {}
-
-            global_yaml[module.name] = CommentedMap()
-            global_yaml.yaml_set_comment_before_after_key(
-                module.name, f"\n\n{module.display_name} configuration options"
-            )
-
-            for key, value in manifest["configs"].items():
-                type = value.get("type", "string")
-                if type == "json_loader":
-                    value["type"] = "json"
-                elif type == "str":
-                    type = "string"
-
-                default = value.get("default", "")
-                config_yaml[key] = default
-
-                global_yaml[module.name][key] = default
-
-                if value.get("help", ""):
-                    global_yaml[module.name].yaml_add_eol_comment(value.get("help", ""), key)
-
-                help = "**Required**. " if value.get("required", False) else "Optional. "
-                help += value.get("help", "")
-                config_table += f"| `{module.name}.{key}` | {help} | {value.get('default', '')} | {type} |\n"
-                global_table += f"| `{module.name}.{key}` | {help} | {default} | {type} |\n"
-            readme_str += "\n## Configuration Options\n"
-            readme_str += "\n### YAML\n"
-
-            config_string = io.BytesIO()
-            yaml.dump({module.name: config_yaml}, config_string)
-            config_string = config_string.getvalue().decode("utf-8")
-        yaml_string = EXAMPLE_YAML.format(steps_str=steps_str, config_string=config_string)
-        readme_str += f"```{{code}} yaml\n{yaml_string}\n```\n"
-
-        if manifest["configs"]:
-            readme_str += "\n### Command Line:\n"
-            readme_str += config_table
-
-        # add a link to the autodoc refs
-        readme_str += f"\n[API Reference](../../../autoapi/{module.name}/index)\n"
-        # create the module.type folder, use the first type just for where to store the file
-        for type in manifest["type"]:
-            type_folder = SAVE_FOLDER / type
-            type_folder.mkdir(exist_ok=True)
-            with open(type_folder / f"{module.name}.md", "w") as f:
-                print("writing", SAVE_FOLDER)
-                f.write(readme_str)
-        generate_index(modules_by_type)
-
-    del global_yaml["placeholder"]
-    global_string = io.BytesIO()
-    global_yaml = yaml.dump(global_yaml, global_string)
-    global_string = global_string.getvalue().decode("utf-8")
-    global_yaml = f"```yaml\n{global_string}\n```"
-    with open(SAVE_FOLDER / "configs_cheatsheet.md", "w") as f:
-        f.write("### Configuration File\n" + global_yaml + "\n### Command Line\n" + global_table)
-
-
-def generate_index(modules_by_type):
-    readme_str = ""
-    for type in BaseModule.MODULE_TYPES:
-        modules = modules_by_type.get(type, [])
-        module_str = f"## {type.capitalize()} Modules\n"
-        for module in modules:
-            module_str += f"\n[{module.manifest['name']}](/modules/autogen/{module.type[0]}/{module.name}.md)\n"
-        with open(SAVE_FOLDER / f"{type}.md", "w") as f:
-            print("writing", SAVE_FOLDER / f"{type}.md")
-            f.write(module_str)
-        readme_str += module_str
-
-    with open(SAVE_FOLDER / "module_list.md", "w") as f:
-        print("writing", SAVE_FOLDER / "module_list.md")
-        f.write(readme_str)
-
-
-if __name__ == "__main__":
-    generate_module_docs()
--- a/docs/source/bc.png
+++ b/docs/source/bc.png
--- a/docs/source/conf.py
+++ b/docs/source/conf.py
@@ -1,94 +0,0 @@
-# Configuration file for the Sphinx documentation builder.
-# https://www.sphinx-doc.org/en/master/usage/configuration.html
-import sys
-import os
-from importlib.metadata import metadata
-from datetime import datetime
-
-sys.path.append(os.path.abspath("../scripts"))
-from scripts import generate_module_docs
-from auto_archiver.version import __version__
-
-# -- Project Hooks -----------------------------------------------------------
-# convert the module __manifest__.py files into markdown files
-generate_module_docs()
-
-
-# -- Project information -----------------------------------------------------
-package_metadata = metadata("auto-archiver")
-project = package_metadata["name"]
-copyright = str(datetime.now().year)
-author = "Bellingcat"
-release = package_metadata["version"]
-language = "en"
-
-# -- General configuration ---------------------------------------------------
-extensions = [
-    "myst_parser",  # Markdown support
-    "autoapi.extension",  # Generate API documentation from docstrings
-    "sphinxcontrib.mermaid",  # Mermaid diagrams
-    "sphinx.ext.viewcode",  # Source code links
-    "sphinx_copybutton",
-    "sphinx.ext.napoleon",  # Google-style and NumPy-style docstrings
-    "sphinx.ext.autosectionlabel",
-    # 'sphinx.ext.autosummary',       # Summarize module/class/function docs
-]
-
-templates_path = ["_templates"]
-exclude_patterns = ["_build", "Thumbs.db", ".DS_Store", ""]
-
-
-# -- AutoAPI Configuration ---------------------------------------------------
-autoapi_type = "python"
-autoapi_dirs = ["../../src/auto_archiver/core/", "../../src/auto_archiver/utils/"]
-# get all the modules and add them to the autoapi_dirs
-autoapi_dirs.extend([f"../../src/auto_archiver/modules/{m}" for m in os.listdir("../../src/auto_archiver/modules")])
-autodoc_typehints = "signature"  # Include type hints in the signature
-autoapi_ignore = [
-    "*/version.py",
-]  # Ignore specific modules
-autoapi_keep_files = True  # Option to retain intermediate JSON files for debugging
-autoapi_add_toctree_entry = True  # Include API docs in the TOC
-autoapi_python_use_implicit_namespaces = True
-autoapi_template_dir = "../_templates/autoapi"
-autoapi_options = [
-    "members",
-    "undoc-members",
-    "show-inheritance",
-    "imported-members",
-]
-
-
-# -- Markdown Support --------------------------------------------------------
-myst_enable_extensions = [
-    "deflist",  # Definition lists
-    "html_admonition",  # HTML-style admonitions
-    "html_image",  # Inline HTML images
-    "replacements",  # Substitutions like (C)
-    "smartquotes",  # Smart quotes
-    "linkify",  # Auto-detect links
-    "substitution",  # Text substitutions
-]
-myst_heading_anchors = 2
-myst_fence_as_directive = ["mermaid"]
-
-source_suffix = {
-    ".rst": "restructuredtext",
-    ".md": "markdown",
-}
-
-# -- Options for HTML output -------------------------------------------------
-html_theme = "sphinx_book_theme"
-html_static_path = ["../_static"]
-html_css_files = ["custom.css"]
-html_title = f"Auto Archiver v{__version__}"
-html_logo = "bc.png"
-html_theme_options = {
-    "repository_url": "https://github.com/bellingcat/auto-archiver",
-    "use_repository_button": True,
-}
-
-
-copybutton_prompt_text = r">>> |\.\.\."
-copybutton_prompt_is_regexp = True
-copybutton_only_copy_prompt_lines = False
--- a/docs/source/contributing.md
+++ b/docs/source/contributing.md
@@ -1,2 +0,0 @@
-```{include} ../../CONTRIBUTING.md
-```
--- a/docs/source/core_modules.md
+++ b/docs/source/core_modules.md
@@ -1,28 +0,0 @@
-# Module Documentation
-
-These pages describe the core modules that come with Auto Archiver and provide the main functionality for archiving websites on the internet. There are five core module types:
-
-1. Feeders - these 'feed' information (the URLs) from various sources to the Auto Archiver for processing
-2. Extractors - these 'extract' the page data for a given URL that is fed in by a feeder
-3. Enrichers - these 'enrich' the data extracted in the previous step with additional information
-4. Storage - these 'store' the data in a persistent location (on disk, Google Drive etc.)
-5. Databases - these 'store' the status of the entire archiving process in a log file or database.
-
-
-```{include} modules/autogen/module_list.md
-```
-
-
-```{toctree}
-:maxdepth: 1
-:caption: Core Modules
-:hidden:
-
-modules/config_cheatsheet
-modules/feeder
-modules/extractor
-modules/enricher
-modules/storage
-modules/database
-modules/formatter
-```
--- a/docs/source/development/creating_modules.md
+++ b/docs/source/development/creating_modules.md
@@ -1,52 +0,0 @@
-# Creating Your Own Modules
-
-Modules are what's used to extend Auto Archiver to process different websites or media, and/or transform the data in a way that suits your needs. In most cases, the [Core Modules](../core_modules.md) should be sufficient for every day use, but the most common use-cases for making your own Modules include:
-
-1. Extracting data from a website which doesn't work with the current core extractors.
-2. Enriching or altering the data before saving with additional information that the core enrichers do not offer.
-3. Storing your data in a different format/location from what the core storage providers offer.
-
-## Setting up the folder structure
-
-1. First, decide what type of module you wish to create. Check the types of modules on the [](../core_modules.md) page to decide what type you need. (Note: a module can be more than one type, more on that below)
-2. Create a new python package (a folder) with the name of your module (in this tutorial, we'll call it `awesome_extractor`).
-3. Create the `__manifest__.py` and an the `awesome_extractor.py` files in this folder.
-
-When done, you should have a module structure as follows:
-
-```
-.
-├── awesome_extractor
-│   ├── __manifest__.py
-│   └── awesome_extractor.py
-``` 
-
-Check out the [core modules](https://github.com/bellingcat/auto-archiver/tree/main/src/auto_archiver/modules) in the Auto Archiver repository for examples of the folder structure for real-world modules.
-
-## Populating the Manifest File
-
-The manifest file is where you define the core information of your module. It is a python dict containing important information, here's an example file:
-
-```{include} ../../../tests/data/test_modules/example_module/__manifest__.py
-:name: __manifest__.py
-:literal:
-:parser: python
-```
-
-## Creating the Python Code
-
-The next step is to create your module code. First, create a class which should subclass the base module types from `auto_archiver.core`, here's an example class for the `awesome_extractor` module which is an `extractor`:
-
-```{code-block} python
-:filename: awesome_extractor.py
-
-from auto_archiver.core import Extractor, Metadata
-
-def AwesomeExtractor(Extractor):
-
-    def download(self, item: Metadata) -> Metadata | False:
-      url = item.get_url()
-      # download the content and create the metadata object
-      metadata = ...
-      return metadata
-```
--- a/docs/source/development/developer_guidelines.md
+++ b/docs/source/development/developer_guidelines.md
@@ -1,36 +0,0 @@
-
-# Developer Guidelines
-
-This section of the documentation provides guidelines for developers who want to modify or contribute to the tool.
-
-
-## Developer Install
-
-1. Clone the project using `git clone https://github.com/bellingcat/auto-archiver.git` 
-2. Install poetry using `curl -sSL https://install.python-poetry.org | python3 -` ([other installation methods](https://python-poetry.org/docs/#installation))
-3. Install dependencies with `poetry install`
-
-## Running 
-4. Run the code with `poetry run auto-archiver [my args]`
-
-```{note}
-Add the plugin [poetry-shell-plugin](https://github.com/python-poetry/poetry-plugin-shell) and run `poetry shell` to activate the virtual environment.
-This allows you to run the auto-archiver without the `poetry run` prefix.
-```
-
-### Optional Development Packages
-
-Install development packages (used for unit tests etc.) using:
-`poetry install -with dev`
-
-
-```{toctree}
-:hidden:
-creating_modules
-docker_development
-testing
-docs
-release
-settings_page
-style_guide
-```
--- a/docs/source/development/docker_development.md
+++ b/docs/source/development/docker_development.md
@@ -1,5 +0,0 @@
-## Docker development
-working with docker locally:
-  * `docker compose up` to build the first time and run a local image with the settings in `secrets/orchestration.yaml`
-  * To modify/pass additional command line args, use `docker compose run auto-archiver --config secrets/orchestration.yaml [OTHER ARGUMENTS]`
-  * To rebuild after code changes, just pass the `--build` flag, e.g. `docker compose up --build`
--- a/docs/source/development/docs.md
+++ b/docs/source/development/docs.md
@@ -1,47 +0,0 @@
-
-### Building the Docs
-
-The documentation is built using [Sphinx](https://www.sphinx-doc.org/en/master/) and [AutoAPI](https://sphinx-autoapi.readthedocs.io/en/latest/) and hosted on ReadTheDocs.
-To build the documentation locally, run the following commands:
-
-**Install required dependencies:**
- Install the docs group of dependencies: 
-```shell
-# only the docs dependencies
-poetry install --only docs
-
-# or for all dependencies 
-poetry install
-```
- Either use [poetry-plugin-shell](https://github.com/python-poetry/poetry-plugin-shell) to activate the virtual environment: `poetry shell`
- Or prepend the following commands with `poetry run`
-
-**Create the documentation:**
- Build the documentation: 
-```shell
-# Using makefile (Linux/macOS):
-make -C docs html
-
-# or using sphinx directly (Windows/Linux/macOS):
-sphinx-build -b html docs/source docs/_build/html
-```
- If you make significant changes and want a fresh build run: `make -C docs clean` to remove the old build files.
-
-**Viewing the documentation:**
-```shell
-# to open the documentation in your browser.
-open docs/_build/html/index.html
-
-# or run autobuild to automatically update the documentation when you make changes
-sphinx-autobuild docs/source docs/_build/html
-```
-
-
-### Managing Readthedocs (RTD) Versions
-
-Version management is done at [https://app.readthedocs.org/projects/auto-archiver/](https://app.readthedocs.org/projects/auto-archiver/)
-(login required). Once logged in, you can create new versions, delete old versions or change visibility of versions. More info on
-[RTD](https://docs.readthedocs.com/platform/stable/versions.html).
-
-Currently, the Auto Archiver project is set up to automatically create a new docs version for each `vX.Y.Z` release. For more on this,
-see the RTD [instructions on automation](https://docs.readthedocs.com/platform/stable/guides/automation-rules.html) or edit the existing automation rule in the project settings.
--- a/docs/source/development/release.md
+++ b/docs/source/development/release.md
@@ -1,33 +0,0 @@
-# Release Process
-
-```{note} This is a work in progress.
-```
-### Update the project version
-
-Update the version number in the project file: [pyproject.toml](../../pyproject.toml) following SemVer:
-```toml
-[project]
-name = "auto-archiver"
-version = "0.1.1"
-```
-Then commit and push the changes.
-
-* The package version is automatically updated in PyPi using the workflow [python-publish.yml](../../.github/workflows/python-publish.yml)
-* A Docker image is automatically pushed with the git tag to dockerhub using the workflow [docker-publish.yml](../../.github/workflows/docker-publish.yml)
-
-### Create the release on Git
-
-The release needs a git tag which should match the project version number, prefixed with a 'v'. For example, if the project version is `0.1.1`, the git tag should be `v0.1.1`.
-This can be done the usual way, or created within the Github UI when you create the release.
-
-Go to GitHub releases > new release > create the release with the new tag and the release notes.
-
-
-manual release to docker hub
-  * `docker image tag auto-archiver bellingcat/auto-archiver:latest`
-  * `docker push bellingcat/auto-archiver`
-
-
-### Building the Settings Page
-
-The Settings page is built as part of the python-publish workflow and packaged within the app.
--- a/docs/source/development/settings_page.md
+++ b/docs/source/development/settings_page.md
@@ -1,31 +0,0 @@
-# Configuration Editor
-
-The [configuration editor](../installation/config_editor.md), is an easy-to-use UI for users to edit their auto-archiver settings.
-
-The single-file app is built using React and vite. To get started developing the package, follow these steps:
-
-1. Make sure you have Node v22 installed.
-
-```{note} Tip: if you don't have node installed:
-
-Use `nvm` to manage your node installations. Use: 
-`curl -o- https://raw.githubusercontent.com/nvm-sh/nvm/v0.40.1/install.sh | bash` to install `nvm` and then `nvm i 22` to install Node v22
-```
-
-2. Generate the `schema.json` file for the currently installed modules using `python scripts/generate_settings_schema.py`
-3. Go to the settings folder `cd scripts/settings/` and build your environment with `npm i`
-4. Run a development version of the page with `npm run dev` and then open localhost:5173.
-5. Build a release version of the page with `npm run build`
-
-A release version creates a single-file app called `dist/index.html`. This file should be copied to `docs/source/installation/settings_base.html` so that it can be integrated into the sphinx docs.
-
-```{note}
-
-The single-file app dist/index.html does not include any `<html>` or `<head>` tags as it is designed to be built into a RTD docs page. Edit `index.html` in the settings folder if you wish to modify the built page.
-```
-
-## Readthedocs Integration
-
-The configuration editor is built as part of the RTD deployment (see `.readthedocs.yaml` file). This command is run every time RTD is built:
-
-`cd scripts/settings && npm install && npm run build && yes | cp dist/index.html ../../docs/source/installation/settings_base.html && cd ../..`
--- a/docs/source/development/style_guide.md
+++ b/docs/source/development/style_guide.md
@@ -1,70 +0,0 @@
-# Style Guide
-
-
-The project uses [Ruff](https://docs.astral.sh/ruff/) for linting and  formatting.
-Our style configurations are set in the `pyproject.toml` file. If needed, you can modify them there.
-
-
-### **Formatting (Auto-Run Before Commit) 🛠️**  
-
-We have a pre-commit hook to run the formatter before you commit.
-This requires you to set it up once locally, then it will run automatically when you commit changes.
-
-```shell
-poetry run pre-commit install
-```
-
-Ruff can also be to run automatically.
-Alternative: Ruff can also be [integrated with most editors](https://docs.astral.sh/ruff/editors/setup/)  for real-time formatting.
-
-If you wish to disable the pre-commit hook (for example, if you want to commit some WIP code) you can use the `--no-verify` flag when you commit.
-For example: `git commit -m "WIP Code" --no-verify`
-
-### **Linting (Check Before Pushing) 🔍**
-
-We recommend you also run the linter before pushing code. 
-
-We have [Makefile](../../../Makefile) commands to run common tasks. 
-
-Tip: if you're on Windows you might need to install `make` first, or alternatively you can use ruff commands directly.
-
-
-**Lint Check:** This outputs a report of any issues found, without attempting to fix them: 
-```shell
-make ruff-check
-```
-
-Tip: To see a more detailed linting report, you can remove the following line from the `pyproject.toml` file:
-```toml
-[tool.ruff]
-
-# Remove this for a more detailed lint report
-output-format = "concise"
-```
-
-**Lint Fix:** This command will attempt to fix some of the issues it picked up with the lint check.
-
-Note not all warnings can be fixed automatically.
-
-⚠️ Warning: This can cause breaking changes. ⚠️
-
-Most fixes are safe, but some non-standard practices such as dynamic loading are not picked up by linters. Ensure you check any modifications by this before committing them.
-```shell
-make ruff-fix
-```
-
-**Changing Configurations ⚙️**
-
-
-Our rules are quite lenient for general usage, but if you want to run more rigorous checks you can then run checks with additional rules to see more nuanced errors which you can review manually.
-Check out the [ruff documentation](https://docs.astral.sh/ruff/configuration/) for the full list of rules.
-One example is to extend the selected rules for linting the `pyproject.toml` file:
-
-```toml 
-[tool.ruff.lint]
-# Extend the rules to check for by adding them to this option:
-# See documentation for more details: https://docs.astral.sh/ruff/rules/
-extend-select = ["B"]
-```
-
-Then re-run the `make ruff-check` command to see the new rules in action.
--- a/docs/source/development/testing.md
+++ b/docs/source/development/testing.md
@@ -1,21 +0,0 @@
-# Testing
-
-`pytest` is used for testing. There are two main types of tests:
-
-1. 'core' tests which should be run on every change
-2. 'download' tests which hit the network. These tests will do things like make API calls (e.g. Twitter, Bluesky etc.) and should be run regularly to make sure that APIs have not changed.
-
-
-## Running Tests 
-
-1. Make sure you've installed the dev dependencies with `pytest install --with dev`
-2. Tests can be run as follows:
-```
-#### Command prefix of 'poetry run' removed here for simplicity
-# run core tests
-pytest -ra -v -m "not download"
-# run download tests
-pytest -ra -v -m "download"
-# run all tests
-pytest -ra -v
-```
--- a/docs/source/example.orchestration.yaml
+++ b/docs/source/example.orchestration.yaml
@@ -1,79 +0,0 @@
-# Auto Archiver Configuration
-# Steps are the modules that will be run in the order they are defined
-
-steps:
-  feeders:
-  - cli_feeder
-  extractors:
-  - generic_extractor
-  - telegram_extractor
-  enrichers:
-  - thumbnail_enricher
-  - meta_enricher
-  - pdq_hash_enricher
-  - ssl_enricher
-  - hash_enricher
-  databases:
-  - console_db
-  - csv_db
-  storages:
-  - local_storage
-  formatters:
-  - html_formatter
-
-# Global configuration
-
-# Authentication
-# a dictionary of authentication information that can be used by extractors to login to website. 
-# you can use a comma separated list for multiple domains on the same line (common usecase: x.com,twitter.com)
-# Common login 'types' are username/password, cookie, api key/token.
-# There are two special keys for using cookies, they are: cookies_file and cookies_from_browser. 
-# Some Examples:
-# facebook.com:
-#   username: "my_username"
-#   password: "my_password"
-# or for a site that uses an API key:
-# twitter.com,x.com:
-#   api_key
-#   api_secret
-# youtube.com:
-#   cookie: "login_cookie=value ; other_cookie=123" # multiple 'key=value' pairs should be separated by ;
-
-authentication: {}
-
-# Logging settings for your project. See the logging settings with --help
-
-logging:
-  level: INFO
-
-# These are the global configurations that are used by the modules
-
-  file:
-  rotation:
-local_storage:
-  path_generator: flat
-  filename_generator: static
-  save_to: ./local_archive
-  save_absolute: false
-html_formatter:
-  detect_thumbnails: true
-thumbnail_enricher:
-  thumbnails_per_minute: 60
-  max_thumbnails: 16
-generic_extractor:
-  subtitles: true
-  comments: false
-  livestreams: false
-  live_from_start: false
-  proxy: ''
-  end_means_success: true
-  allow_playlist: false
-  max_downloads: inf
-csv_db:
-  csv_file: db.csv
-ssl_enricher:
-  skip_when_nothing_archived: true
-hash_enricher:
-  algorithm: SHA-256
-  chunksize: 16000000
-
--- a/docs/source/flow_overview.md
+++ b/docs/source/flow_overview.md
@@ -1,30 +0,0 @@
-
-# Archiving Overview
-
-The archiver archives web pages using the following workflow
-1. **Feeder** gets the links (from a spreadsheet, from the console, ...)
-2. **Extractor** tries to extract content from the given link (e.g. videos from youtube, images from Twitter...)
-3. **Enricher** adds more info to the content (hashes, thumbnails, ...)
-4. **Formatter** creates a report from all the archived content (HTML, PDF, ...)
-5. **Database** knows what's been archived and also stores the archive result (spreadsheet, CSV, or just the console)
-
-Each step in the workflow is handled by 'modules' that interact with the data in different ways. For example, the Twitter Extractor Module would extract information from the Twitter website. The Screenshot Enricher Module will take screenshots of the given page. See the [core modules page](core_modules.md) for an overview of all the modules that are available.
-
-Auto-archiver must have at least one module defined for each step of the workflow. This is done by setting the [configuration](installation/configurations.md) for your auto-archiver instance.
-
-Here's the complete workflow that the auto-archiver goes through:
-
-```mermaid
-
-graph TD
-    s((start)) --> F(fa:fa-table Feeder)
-    F -->|get and clean URL| D1{fa:fa-database Database}
-    D1 -->|is already archived| e((end))
-    D1 -->|not yet archived| a(fa:fa-download Archivers)
-    a -->|got media| E(fa:fa-chart-line Enrichers)
-    E --> S[fa:fa-box-archive Storages]
-    E --> Fo(fa:fa-code Formatter)
-    Fo --> S
-    Fo -->|update database| D2(fa:fa-database Database)
-    D2 --> e
-```
--- a/docs/source/how_to.md
+++ b/docs/source/how_to.md
@@ -1,12 +0,0 @@
-# How-To Guides
-
-The follow pages contain helpful how-to guides for common use cases of the Auto Archiver.
---
-
-```{toctree}
-:maxdepth: 1
-:glob:
-
-how_to/*
-
-```
--- a/docs/source/how_to/authentication_how_to.md
+++ b/docs/source/how_to/authentication_how_to.md
@@ -1,110 +0,0 @@
-# Logging in to sites
-
-This how-to guide shows you how you can use various authentication methods to allow you to login to a site you are trying to archive. This is useful for websites that require a user to be logged in to browse them, or for sites that restrict bots.
-
-In this How-To, we will authenticate on use Twitter/X.com using cookies, and on XXXX using username/password.
-
-
-
-## Using cookies to authenticate on Twitter/X
-
-It can be useful to archive tweets after logging in, since some tweets are only visible to authenticated users. One case is Tweets marked as 'Sensitive'.
-
-Take this tweet as an example: [https://x.com/SozinhoRamalho/status/1876710769913450647](https://x.com/SozinhoRamalho/status/1876710769913450647)
-
-This tweet has been marked as sensitive, so a normal run of Auto Archiver without a logged in session will fail to extract the tweet:
-
-```{code-block} console
-:emphasize-lines: 3,4,5,6
-
->>> auto-archiver https://x.com/SozinhoRamalho/status/1876710769913450647                                                                                     ✭ ✱
- ...
-ERROR: [twitter] 1876710769913450647: NSFW tweet requires authentication. Use --cookies, 
--cookies-from-browser, --username and --password, --netrc-cmd, or --netrc (twitter) to
- provide account credentials. See https://github.com/yt-dlp/yt-dlp/wiki/FAQ#how-do-i-pass-cookies-to-yt-dlp 
- for how to manually pass cookies
-[twitter] 1876710769913450647: Downloading guest token
-[twitter] 1876710769913450647: Downloading GraphQL JSON
-2025-02-20 15:06:13.362 | ERROR    | auto_archiver.modules.generic_extractor.generic_extractor:download_for_extractor:248 - Error downloading metadata for post: NSFW tweet requires authentication. Use --cookies, --cookies-from-browser, --username and --password, --netrc-cmd, or --netrc (twitter) to provide account credentials. See  https://github.com/yt-dlp/yt-dlp/wiki/FAQ#how-do-i-pass-cookies-to-yt-dlp  for how to manually pass cookies
-[generic] Extracting URL: https://x.com/SozinhoRamalho/status/1876710769913450647
-[generic] 1876710769913450647: Downloading webpage
-WARNING: [generic] Falling back on generic information extractor
-[generic] 1876710769913450647: Extracting information
-ERROR: Unsupported URL: https://x.com/SozinhoRamalho/status/1876710769913450647
-2025-02-20 15:06:13.744 | INFO     | auto_archiver.core.orchestrator:archive:483 - Trying extractor telegram_extractor for https://x.com/SozinhoRamalho/status/1876710769913450647
-2025-02-20 15:06:13.744 | SUCCESS  | auto_archiver.modules.console_db.console_db:done:23 - DONE Metadata(status='nothing archived', metadata={'_processed_at': datetime.datetime(2025, 2, 20, 15, 6, 12, 473979, tzinfo=datetime.timezone.utc), 'url': 'https://x.com/SozinhoRamalho/status/1876710769913450647'}, media=[])
-...
-```
-
-To get round this limitation, we can use **cookies** (information about a logged in user) to mimic being logged in to Twitter. There are two ways to pass cookies to Auto Archiver. One is from a file, and the other is from a browser profile on your computer.
-
-In this tutorial, we will export the Twitter cookies from our browser and add them to Auto Archiver
-
-**1. Installing a cookie exporter extension**
-
-First, we need to install an extension in our browser to export the cookies for a certain site. The [FAQ on yt-dlp](https://github.com/yt-dlp/yt-dlp/wiki/FAQ#how-do-i-pass-cookies-to-yt-dlp) provides some suggestions: Get [cookies.txt LOCALLY](https://chromewebstore.google.com/detail/get-cookiestxt-locally/cclelndahbckbenkjhflpdbgdldlbecc) for Chrome or [cookies.txt](https://addons.mozilla.org/en-US/firefox/addon/cookies-txt/) for Firefox.
-
-**2. Export the cookies**
-
-```{note} See the note [here](../installation/authentication.md#recommendations-for-authentication) on why you shouldn't use your own personal account for archiving.
-```
-
-Once the extension is installed in your preferred browser, login to Twitter in this browser, and then activate the extension and export the cookies. You can choose to export all your cookies for your browser, or just cookies for this specific site. In the image below, we're only exporting cookies for Twitter/x.com:
-
-![extract cookies](extract_cookies.png)
-
-
-**3. Adding the cookies file to Auto Archiver**
-
-You now will have a file called `cookies.txt` (tip: name it `twitter_cookies.txt` if you only exported cookies for Twitter), which needs to be added to Auto Archiver.
-
-Do this by going into your Auto Archiver configuration file, and editing the `authentication` section. We will add the `cookies_file` option for the site `x.com,twitter.com`.
-
-```{note} For websites that have multiple URLs (like x.com and twitter.com) you can 'reuse' the same login information without duplicating it using a comma separated list of domain names.
-```
-
-I've saved my `twitter_cookies.txt` file in a `secrets` folder, so here's how my authentication section looks now:
-
-```{code} yaml
-:caption: orchestration.yaml
-
-...
-
-authentication:
-   x.com,twitter.com:
-      cookies_file: secrets/twitter_cookies.txt
-...
-```
-
-**4. Re-run your archiving with the cookies enabled**
-
-Now, the next time we re-run Auto Archiver, the cookies from our logged-in session will be used by Auto Archiver, and restricted/sensitive tweets can be downloaded!
-
-```{code} console
->>> auto-archiver https://x.com/SozinhoRamalho/status/1876710769913450647                                                                                   ✭ ✱ ◼
-...
-2025-02-20 15:27:46.785 | WARNING  | auto_archiver.modules.console_db.console_db:started:13 - STARTED Metadata(status='no archiver', metadata={'_processed_at': datetime.datetime(2025, 2, 20, 15, 27, 46, 785304, tzinfo=datetime.timezone.utc), 'url': 'https://x.com/SozinhoRamalho/status/1876710769913450647'}, media=[])
-2025-02-20 15:27:46.785 | INFO     | auto_archiver.core.orchestrator:archive:483 - Trying extractor generic_extractor for https://x.com/SozinhoRamalho/status/1876710769913450647
-[twitter] Extracting URL: https://x.com/SozinhoRamalho/status/1876710769913450647
-...
-2025-02-20 15:27:53.134 | INFO     | auto_archiver.modules.local_storage.local_storage:upload:26 - ./local_archive/https-x-com-sozinhoramalho-status-1876710769913450647/06e8bacf27ac4bb983bf6280.html
-2025-02-20 15:27:53.135 | SUCCESS  | auto_archiver.modules.console_db.console_db:done:23 - DONE Metadata(status='yt-dlp_Twitter: success', 
-metadata={'_processed_at': datetime.datetime(2025, 2, 20, 15, 27, 48, 564738, tzinfo=datetime.timezone.utc), 'url': 
-'https://x.com/SozinhoRamalho/status/1876710769913450647', 'title': 'ignore tweet, testing sensitivity warning nudity https://t.co/t3u0hQsSB1', 
-...
-```
-
-
-### Finishing Touches
-
-You've now successfully exported your cookies from a logged-in session in your browser, and used them to authenticate with Twitter and download a sensitive tweet. Congratulations!
-
-Finally,Some important things to remember:
-
-1. It's best not to use your own personal account for archiving. [Here's why](../installation/authentication.md#recommendations-for-authentication).
-2. Cookies can be short-lived, so may need updating. Sometimes, a website session may 'expire' or a website may force you to login again. In these instances, you'll need to repeat the export step (step 2) after logging in again to update your cookies.
-
-## Authenticating on XXXX site with username/password
-
-```{note} This section is still under construction 🚧
-```
--- a/docs/source/how_to/extract_cookies.png
+++ b/docs/source/how_to/extract_cookies.png
--- a/docs/source/how_to/gsheets_setup.md
+++ b/docs/source/how_to/gsheets_setup.md
@@ -1,159 +0,0 @@
-# Using Google Sheets
-
-This guide explains how to set up Google Sheets to process URLs automatically and then store the archiving status back into the Google sheet. It is broadly split into 3 steps:
-
-1. Setting up your Google Sheet
-2. Setting up a service account so Auto Archiver can access the sheet
-3. Setting the Auto Archiver settings
-
-### 1. Setting up your Google Sheet
-
-Any Google sheet must have at least *one* column, with the name 'link' (you can change this name afterwards). This is the column with the URLs that you want the Auto Archiver to archive. 
-Your sheet can have many other columns that the Auto Archiver can use, and you can also include any additional columns for your own personal use. The order of the columns does not matter, the naming just needs to be correctly assigned to its corresponding value in the configuration file.
-
-We recommend copying [this template Google Sheet](https://docs.google.com/spreadsheets/d/1NJZo_XZUBKTI1Ghlgi4nTPVvCfb0HXAs6j5tNGas72k/edit?usp=sharing) as a starting point for your project, as this matches the default column names.
-
-Here's an overview of all the columns, and what a complete sheet would look like.
-
-**Inputs:**
-
-These are processed by the Gsheet Feeder and passed to the Auto Archiver.
-
-* **Link** *(required)*: the URL of the post that is to be archived
-* **Destination folder**: custom folder for archived file (regardless of storage)
-
-**Outputs:**
-
-These are updated by the Gsheet DB module during the archiving process.
-Note the required columns are only required if you are using the Gsheet DB module as well as the feeder.
-
-* **Archive status** *(required)*: Status of archive operation
-* **Archive location**: URL of archived post
-* **Archive date**: Date archived
-* **Thumbnail**: Embeds a thumbnail for the post in the spreadsheet
-* **Timestamp**: Timestamp of original post
-* **Title**: Post title
-* **Text**: Post text
-* **Screenshot**: Link to screenshot of post
-* **Hash**: Hash of archived HTML file (which contains hashes of post media) - for checksums/verification
-* **Perceptual Hash**: Perceptual hashes of found images - these can be used for de-duplication of content
-* **WACZ**: Link to a WACZ web archive of post
-* **ReplayWebpage**: Link to a ReplayWebpage viewer of the WACZ archive
-
-For example, this is a spreadsheet configured with all of the columns for the auto archiver and a few URLs to archive. 
-In this example the Ghseet Feeder and Gsheet DB are being used, and the archive is in progress.
-(Note that the column names are not case sensitive.)
-
-![A screenshot of a Google Spreadsheet with column headers defined as above, and several Youtube and Twitter URLs in the "Link" column](../../demo-before.png)
-
-We'll change the name of the 'Destination Folder' column in step 3.
-
-## 2. Setting up your Service Account
-
-Once your Google Sheet is set up, you need to create what's called a 'service account' that will allow the Auto Archiver to access it.
-
-To do this, follow the steps in [this guide](https://gspread.readthedocs.io/en/latest/oauth2.html) all the way up until step 8. You should have downloaded a file called `service_account.json` and shared the Google Sheet with the log 'client_email' email address in this file.
-
-Once you've downloaded the file, save it to `secrets/service_account.json`
-
-## 3. Setting up the configuration file
-
-Now that you've set up your Google sheet, and you've set up the service account so Auto Archiver can access the sheet, the final step is to set your configuration.
-
-First, make sure you have `gsheet_feeder_db` set in the `steps.feeders` section of your config. If you wish to store the results of the archiving process back in your Google sheet, make sure to also set the `ghseet_db` settig in the `steps.databases` section. Here's how this might look:
-
-```{code} yaml
-steps:
-    feeders:
-    - gsheet_feeder_db
-    ...
-    databases:
-    - gsheet_feeder_db # optional, if you also want to store the results in the Google sheet and tract the status of active archivals.
-    ...
-```
-
-Next, set up the `gsheet_feeder_db` configuration settings in the 'Configurations' part of the config `orchestration.yaml` file. Open up the file, and set the `gsheet_feeder_db.sheet` setting or the `gsheet_feeder_db.sheet_id` setting. The `sheet` should be the name of your sheet, as it shows in the top left of the sheet. 
-For example, the sheet [here](https://docs.google.com/spreadsheets/d/1NJZo_XZUBKTI1Ghlgi4nTPVvCfb0HXAs6j5tNGas72k/edit?gid=0#gid=0) is called 'Public Auto Archiver template'.
-
-Here's how this might look:
-
-```{code} yaml
-...
-gsheet_feeder_db:
-    sheet: 'My Awesome Sheet'
-    ...
-```
-
-You can also pass these settings directly on the command line without having to edit the file, here'a an example of how to do that (using docker):
-
-`docker run -it --rm -v $PWD/secrets:/app/secrets -v $PWD/local_archive:/app/local_archive bellingcat/auto-archiver:dockerize --gsheet_feeder_db.sheet "My Awesome Sheet 2"`. 
-
-Here, the sheet name has been overridden/specified in the command line invocation.
-
-### 3a. (Optional) Changing the column names
-
-In step 1, we said we would change the name of the 'Destination Folder'. Perhaps you don't like this name, or already have a sheet with a different name. In our example here, we want to name this column 'Save Folder'. To do this, we need to edit the `ghseet_feeder_db.column` setting in the configuration file. 
-For more information on this setting, see the [Gsheet Feeder Database docs](../modules/autogen/feeder/gsheet_feeder_db.md#configuration-options). We will first copy the default settings from the Gsheet Feeder docs for the 'column' settings, and then edit the 'Destination Folder' section to rename it 'Save Folder'. Our final configuration section looks like:
-
-```{code} yaml
-...
-gsheet_feeder_db:
-    sheet: 'My Awesome Sheet'
-    header: 1
-    service_account: secrets/service_account.json
-    columns:
-      url: link
-      status: archive status
-      folder: save folder # <-- note how this value has been changed
-      archive: archive location
-      date: archive date
-      thumbnail: thumbnail
-      timestamp: upload timestamp
-      title: upload title
-      text: text content
-      screenshot: screenshot
-      hash: hash
-      pdq_hash: perceptual hashes
-      wacz: wacz
-      replaywebpage: replaywebpage
-    
-```
-## 4. Running the Auto Archiver
-### Feeding the URLs to the Auto Archiver
-
-The URLs to be archived should be added to the Google Sheet, and optionally a folder value. Leave all the other configured columns empty (but you may add additional columns for your own use, as long as they don't conflict with the column names mapped in the configuration file).
-The Auto Archiver will archive  any URLs which have an empty 'status' column
-
-### Viewing the Results after archiving
-
-With the `ghseet_feeder_db` installed, once you start running the Auto Archiver, it will update the "Archive status" column.
-The status will be set to "Archive in progress" once the archival starts. If the archival is stopped during a run, either manually or because an error is raised the status value should be cleared.
-
-![A screenshot of a Google Spreadsheet with column headers defined as above, and several Youtube and Twitter URLs in the "Link" column. The auto archiver has added "archive in progress" to one of the status columns.](../../demo-progress.png)
-
-The links are downloaded and archived, and the spreadsheet is updated to the following:
-
-![A screenshot of a Google Spreadsheet with videos archived and metadata added per the description of the columns above.](../../demo-after.png)
-
-Note that the first row is skipped, as it is assumed to be a header row (`--gsheet_feeder_db.header=1` and you can change it if you use more rows above). Rows with an empty URL column, or a non-empty archive column are also skipped. All sheets in the document will be checked.
-
-The "archive location" link contains the path of the archived file, in local storage, S3, or in Google Drive.
-
-![The archive result for a link in the demo sheet.](../../demo-archive.png)
-
-### Troubleshooting
-
-**Hanging Archival in progress status**
-
-Occasionally system crashes or other unexpected events can cause the Auto Archiver to exit without cleaning up the status value.
-If you are sure that all archival processes have stopped but you still see "Archive in progress" in the status column, you can manually clear the status column to allow the Auto Archiver to retry that archival on the next run.
-
-**Nothing archived status**
-
-Sometimes this means the tool is genuinely unable to extract the content at this point in time, but sometimes it can be resolved with different configurations. 
-Try:
-  - Turning on additional 'extractor' types in the configuration file (this can appear as 'no archiver' in the status column). 
-  - Changing credentials or refreshing session files for extractors which require them
-  - Check if the extractors can accept any additional configurations such as adding a cookie file.
-
-
--- a/docs/source/how_to/logging.md
+++ b/docs/source/how_to/logging.md
@@ -1,71 +0,0 @@
-# Keeping Logs
-
-Auto Archiver's logs can be helpful for debugging problematic archiving processes. This guide shows you how to use the logs to 
-
-## Setting up logging
-
-Logging settings can be set on the command line or using the orchestration config file ([learn more](../installation/configuration)). A special `logging` section defines the logging options.
-
-#### Enabling or Disabling Logging
-
-Logging to the console is enabled by default. If you want to globally disable Auto Archiver's logging, then you can set `enabled: false` in your `logging` config:
-
-```{code} yaml
-
-...
-logging:
-   enabled: false
-...
-```
-
-```{note}
-This will disable all logs from Auto Archiver, but it does not disable logs for other tools that the Auto Archiver uses (for example: yt-dlp, firefox or ffmpeg). These logs will still appear in your console.
-```
-
-#### Logging Level
-
-There are 7 logging levels in total, with 4 commonly used levels. They are: `DEBUG`, `INFO`, `WARNING` and `ERROR`.
-
-Change the warning level by setting the value in your orchestration config file:
-
-```{code} yaml
-:caption: orchestration.yaml
-
-...
-logging:
-    level: DEBUG # or INFO / WARNING / ERROR
-...
-```
-
-For normal usage, it is recommended to use the `INFO` level, or if you prefer quieter logs with less information, you can use the `WARNING` level. If you encounter issues with the archiving, then it's recommended to enable the `DEBUG` level.
-
-```{note} To learn about all logging levels, see the [loguru documentation](https://loguru.readthedocs.io/en/stable/api/logger.html)
-```
-
-### Logging to a file
-
-As default, auto-archiver will log to the console. But if you wish to store your logs for future reference, or you are running the auto-archiver from within code a implementation, then you may with to enable file logging. This can be done by setting the `file:` config value in the logging settings.
-
-**Rotation:** For file logging, you can choose to 'rotate' your log files (creating new log files) so they do not get too large. Change this by setting the 'rotation' option in your logging settings. For a full list of rotation options, see the [loguru docs](https://loguru.readthedocs.io/en/stable/overview.html#easier-file-logging-with-rotation-retention-compression).
-
-```{code} yaml
-:caption: orchestration.yaml
-
-logging:
-    ...
-    file: /my/log/file.log
-    rotation: 1 day
-```
-
-### Full logging example
-
-The below example logs only `WARNING` logs to the console and to the file `/my/file.log`, rotating that file once per week:
-
-```{code} yaml
-:caption: orchestration.yaml
-
-logging:
-    level: WARNING
-    file: /my/file.log
-    rotation: 1 week
-```
--- a/docs/source/how_to/new_config_format.md
+++ b/docs/source/how_to/new_config_format.md
@@ -1,146 +0,0 @@
-# Upgrading from v0.12
-
-```{note} This how-to is only relevant for people who used Auto Archiver before February 2025 (versions prior to 0.13).
-
-If you are new to Auto Archiver, then you are already using the latest configuration format and this how-to is not relevant for you.
-```
-
-Versions 0.13+ of Auto Archiver has breaking changes in the configuration format, which means earlier configuration formats will not work without slight modifications.
-
-## How do I know if I need to update my configuration format?
-
-There are two simple ways to check if you need to update your format:
-
-1. When you try and run auto-archiver using your existing configuration file, you get an error about no feeders or formatters being configured, like:
-
-```{code} console
-AssertionError: No feeders were configured. Make sure to set at least one feeder in
-your configuration file or on the command line (using --feeders)
-```
-
-2. Within your configuration file, you have a `feeder:` option. This is the old format. An example old format:
-```{code} yaml
-
-steps:
-  feeder: cli_feeder
-...
-```
-
-The next two sections outline the two methods you have for updating your file.
-
-## 1. Manually edit the configuration file and change the values.
-
-This is recommended if you want to keep all your old settings. Follow the steps below to change the relevant settings:
-
-#### a) Feeder & Formatter Steps Settings
-
-The feeder and formatter settings have been changed from a single string to a list.
-
- `steps.feeder (string)` → `steps.feeders (list)`
- `steps.formatter (string)` → `steps.formatters (list)`
-
-Example:
-
-```{code} yaml
-
-steps:
-   feeder: cli_feeder
-   ...
-   formatter: html_formatter
-
-# the above should be changed to:
-steps:
-   feeders:
-   - cli_feeder
-   ...
-   formatters:
-   - html_formatter
-```
-
-```{note} Auto Archiver still only supports one feeder and formatter, but from v0.13 onwards they must be added to the configuration file as a list.
-```
-
-#### b) Extractor (formerly Archiver) Steps Settings
-
-With v0.13 of Auto Archiver, `archivers` have been renamed to `extractors` to better reflect what they actually do - extract information from a URL. Change the configuration by renaming:
-
- `steps.archivers` → `steps.extractors`
-
-The names of the actual modules have also changed, so for any extractor modules you have enabled, you will need to rename the `archiver` part to `extractor`. Some examples:
-
- `telethon_archiver` → `telethon_extractor`
- `wacz_archiver_enricher` → `wacz_extractor_enricher`
- `wayback_archiver_enricher` → `wayback_extractor_enricher`
- `vk_archiver` → `vk_extractor`
-
-
-#### c) Module Renaming
-
-
-The `youtube_archiver` has been renamed to `generic_extractor` as it is considered the default/fallback extractor. Read more about the [generic extractor](../modules/autogen/extractor/generic_extractor.md).
-
-The `atlos` modules have been merged into one, as have the `gsheets` feeder and database.
-
- `atlos_feeder` → `atlos_feeder_db_storage`
- `atlos_storage` → `atlos_feeder_db_storage`
- `atlos_db` → `atlos_feeder_db_storage`
- `gsheet_feeder` → `gsheet_feeder_db`
- `gsheet_db` → `gsheet_feeder_db`
-
-
-Example:
-```{code} yaml
-steps:
-   feeders:
-   - gsheet_feeder_db # formerly gsheet_feeder
-   ...
-   extractors: # formerly 'archivers'
-   - telethon_extractor # formerly telethon_archiver
-   - generic_extractor # formerly youtube_archiver
-   - vk_extractor # formerly vk_archiver
-   databases:
-   - gsheet_feeder_db # formerly gsheet_db
-   ...
-
-```
-
-```{note}
-
-Don't forget to also rename the configuration settings. For example:
-
-```{code} yaml
-gsheet_feeder_db: # formerly gsheet_feeder
-  service_account: secrets/service_account.json
-  sheet: My Google Sheet
-...
-```
-
-#### d) Redundant / Obsolete Modules
-
-With v0.13 of Auto Archiver, the following modules have been removed and their features have been built in to the generic_extractor. You should remove them from the 'steps' section of your configuration file:
-
-* `twitter_archiver` - use the `generic_extractor` for general extraction, or the `twitter_api_extractor` for API access.
-* `tiktok_archiver` - use the `generic_extractor` to extract TikTok videos.
-
-
-## 2. Auto-generate a new config, then copy over your settings.
-
-Using this method, you can have Auto Archiver auto-generate a configuration file for you, then you can copy over the desired settings from your old config file. This is probably the easiest method and quickest to setup, but it may require some trial and error as you copy over your settings.
-
-First, move your existing `orchestration.yaml` file to a different folder or rename it.
-
-Then, you can generate a `simple` or `full` config using:
-
-```{code} console
->>> # generate a simple config
->>> auto-archiver 
->>> # config will be written to orchestration.yaml
->>> 
->>> # generate a full config
->>> auto-archiver --mode=full
->>> 
-```
-
-After this, copy over any settings from your old config to the new config.
-
-
--- a/docs/source/index.md
+++ b/docs/source/index.md
@@ -1,17 +0,0 @@
-
-```{include} ../../README.md
-```
-
-```{toctree}
-:maxdepth: 2
-:hidden:
-:caption: Contents:
-
-Overview <self>
-installation/setup
-core_modules.md
-how_to
-contributing
-development/developer_guidelines
-autoapi/index.rst
-```
--- a/docs/source/installation/authentication.md
+++ b/docs/source/installation/authentication.md
@@ -1,72 +0,0 @@
-# Authentication
-
-The Authentication framework for auto-archiver allows you to add login details for various websites in a flexible way, directly from the configuration file.
-
-There are two main use cases for authentication:
-* Some websites require some kind of authentication in order to view the content. Examples include Facebook, Telegram etc.
-* Some websites use anti-bot systems to block bot-like tools from accessing the website. Adding real login information to auto-archiver can sometimes bypass this.
-
-## The Authentication Config
-
-You can save your authentication information directly inside your orchestration config file, or as a separate file (for security/multi-deploy purposes). Whether storing your settings inside the orchestration file, or as a separate file, the configuration format is the same. Currently, auto-archiver supports the following authentication types:
-
-**Username & Password:**
- `username`: str - the username to use for login
- `password`: str - the password to use for login
-
-**API**
- `api_key`: str - the API key to use for login
- `api_secret`: str - the API secret to use for login
-  
-**Cookies**
- `cookie`: str - a cookie string to use for login (specific to this site)
- `cookies_from_browser`: str - load cookies from this browser, for this site only.
- `cookies_file`: str - load cookies from this file, for this site only.
-
-```{note} 
-
-The Username & Password, and API settings only work with the Generic Extractor. Other modules (like the screenshot enricher) can only use the `cookies` options. Furthermore, many sites can still detect bots and block username/password logins. Twitter/X and YouTube are two prominent ones that block username/password logging.
-
-One of the 'Cookies' options is recommended for the most robust archiving.
-```
-
-```{code} yaml
-authentication:
-   # optional file to load authentication information from, for security or multi-system deploy purposes
-   load_from_file: path/to/authentication/file.txt
-   # optional setting to load cookies from the named browser on the system, for **ALL** websites
-   cookies_from_browser: firefox
-   # optional setting to load cookies from a cookies.txt/cookies.jar file, for **ALL** websites. See note below on extracting these
-   cookies_file: path/to/cookies.jar
-
-   mysite.com:
-      username: myusername
-      password: 123
-    
-    facebook.com:
-       cookie: single_cookie
-
-    othersite.com:
-       api_key: 123
-       api_secret: 1234
-  
-```
-
-
-### Recommendations for authentication
-
-1. **Store authentication information separately:**
-The authentication part of your configuration contains sensitive information. You should make efforts not to share this with others. For extra security, use the `load_from_file` option to keep your authentication settings out of your configuration file, ideally in a different folder.
-
-2. **Don't use your own personal credentials**
-Depending on the website you are extracting information from, there may be rules (Terms of Service) that prohibit you from scraping or extracting information using a bot. If you use your own personal account, there's a possibility it might get blocked/disabled. It's recommended to set up a separate, 'throwaway' account. In that way, if it gets blocked you can easily create another one to continue your archiving.
-
-
-### How to create a cookies.jar or pass cookies directly to auto-archiver
-
-auto-archiver uses yt-dlp's powerful cookies features under the hood. For instructions on how to extract a cookies.jar (or cookies.txt) file directly from your browser, see the FAQ in the [yt-dlp documentation](https://github.com/yt-dlp/yt-dlp/wiki/FAQ#how-do-i-pass-cookies-to-yt-dlp)
-
-```{note} For developers:
-
-For information on how to access and use authentication settings from within your module, see the `{generic_extractor}` for an example, or view the [`auth_for_site()` function in BaseModule](../autoapi/core/base_module/index.rst)
-```
--- a/docs/source/installation/config_cheatsheet.md
+++ b/docs/source/installation/config_cheatsheet.md
@@ -1,6 +0,0 @@
-# Configuration Cheat Sheet
-
-Below is a list of all configurations for the core modules in Auto Archiver
-
-```{include} ../modules/autogen/configs_cheatsheet.md
-```
--- a/docs/source/installation/config_editor.md
+++ b/docs/source/installation/config_editor.md
@@ -1,5 +0,0 @@
-# Configuration Editor
-
-```{raw} html
-:file: settings.html
-```
--- a/docs/source/installation/configurations.md
+++ b/docs/source/installation/configurations.md
@@ -1,103 +0,0 @@
-
-# Configuration
-
-The recommended way to configure auto-archiver for first-time users is to [run the Auto Archiver](setup.md#running) and have it auto-generate a default configuration for you. Then, if needed, you can edit the configuration file using one of the following methods.
-
-
-## 1. Configuration file
-
-The configuration file is typically called `orchestration.yaml` and stored in the `secrets` folder on your desktop. The configuration file contains all the settings for your entire Auto Archiver workflow in one easy-to-find place.
-
-If you want to have Auto Archiver run with the recommended 'basic' setup, 
-
-### Advanced Configuration
-
-The structure of orchestration file is split into 2 parts: `steps` (what [steps](../flow_overview.md) to use) and `configurations` (settings for individual modules).
-
-A default `orchestration.yaml` will be created for you the first time you run auto-archiver (without any arguments). Here's what it looks like:
-
-<details>
-<summary>View exampleorchestration.yaml</summary>
-
-```{literalinclude} ../example.orchestration.yaml
-   :language: yaml
-   :caption: orchestration.yaml
-```
-
-</details>
-
-## 2. Command Line configuration
-
-You can run auto-archiver directly from the command line, without the need for a configuration file, command line arguments are parsed using the format `module_name.config_value`. For example, a config value of `api_key` in the `instagram_extractor` module would be passed on the command line with the flag `--instagram_extractor.api_key=API_KEY`.
-
-The command line arguments are useful for testing or editing config values and enabling/disabling modules on the fly. When you are happy with your settings, you can store them back in your configuration file by passing the `-s/--store` flag on the command line.
-
-```bash
-auto-archiver --instagram_extractor.api_key=123 --other_module.setting --store
-# will store the new settings into the configuration file (default: orchestration.yaml)
-```
-
-```{note} Arguments passed on the command line override those saved in your settings file. Save them to your config file using the -s or --store flag
-```
-
-## Seeing all Configuration Options
-
-View the configurable settings for the core modules on the individual doc pages for each [](../core_modules.md).
-You can also view all settings available for the modules you have on your system using the `--help` flag in auto-archiver.
-
-```{code-block} console
-:caption: Example output when using the --help flag with auto-archiver
-$ auto-archiver --help
-...
-Positional Arguments:
-  urls                  URL(s) to archive, either a single URL or a list of urls, should not come from config.yaml
-
-Options:
-  --help, -h            show a full help message and exit
-  --version             show program's version number and exit
-  --config CONFIG_FILE  the filename of the YAML configuration file (defaults to 'config.yaml')
-  --mode {simple,full}  the mode to run the archiver in
-  -s, --store, --no-store
-                        Store the created config in the config file
-  --module_paths MODULE_PATHS [MODULE_PATHS ...]
-                        additional paths to search for modules
-  --feeders STEPS.FEEDERS [STEPS.FEEDERS ...]
-                        the feeders to use
-  --enrichers STEPS.ENRICHERS [STEPS.ENRICHERS ...]
-                        the enrichers to use
-  --extractors STEPS.EXTRACTORS [STEPS.EXTRACTORS ...]
-                        the extractors to use
-  --databases STEPS.DATABASES [STEPS.DATABASES ...]
-                        the databases to use
-  --storages STEPS.STORAGES [STEPS.STORAGES ...]
-                        the storages to use
-  --formatters STEPS.FORMATTERS [STEPS.FORMATTERS ...]
-                        the formatter to use
-  --authentication AUTHENTICATION
-                        A dictionary of sites and their authentication methods (token, username etc.) that extractors can use to log into a website. If passing this on the command line, use a JSON string. You may
-                        also pass a path to a valid JSON/YAML file which will be parsed.
-  --logging.level {INFO,DEBUG,ERROR,WARNING}
-                        the logging level to use
-  --logging.file LOGGING.FILE
-                        the logging file to write to
-  --logging.rotation LOGGING.ROTATION
-                        the logging rotation to use
-
-Wayback Machine Enricher:
-  Submits the current URL to the Wayback Machine for archiving and returns either a job ID or the...
-
-  --wayback_extractor_enricher.timeout TIMEOUT
-                        seconds to wait for successful archive confirmation from wayback, if more than this passes the result contains the job_id so the status can later be checked manually.
-  --wayback_extractor_enricher.if_not_archived_within IF_NOT_ARCHIVED_WITHIN
-                        only tell wayback to archive if no archive is available before the number of seconds specified, use None to ignore this option. For more information:
-                        https://docs.google.com/document/d/1Nsv52MvSjbLb2PCpHlat0gkzw0EvtSgpKHu4mk0MnrA
-  --wayback_extractor_enricher.key KEY
-                        wayback API key. to get credentials visit https://archive.org/account/s3.php
-  --wayback_extractor_enricher.secret SECRET
-                        wayback API secret. to get credentials visit https://archive.org/account/s3.php
-  --wayback_extractor_enricher.proxy_http PROXY_HTTP
-                        http proxy to use for wayback requests, eg http://proxy-user:password@proxy-ip:port
-  --wayback_extractor_enricher.proxy_https PROXY_HTTPS
-                        https proxy to use for wayback requests, eg https://proxy-user:password@proxy-ip:port
-```
-
--- a/docs/source/installation/faq.md
+++ b/docs/source/installation/faq.md
@@ -1,60 +0,0 @@
-# Frequently Asked Questions
-
-
-### Q: What websites does the Auto Archiver support?
-**A:** The Auto Archiver works for a large variety of sites. Firstly, the Auto Archiver can download
-and archive any video website supported by YT-DLP, a powerful video-downloading tool ([full list of of
-sites here](https://github.com/yt-dlp/yt-dlp/blob/master/supportedsites.md)). Aside from these sites,
-there are various different 'Extractors' for specific websites. See the full list of extractors that 
-are available on the [extractors](../modules/extractor.md) page. Some sites supported include:
-
-* Twitter
-* Instagram
-* Telegram
-* VKontact
-* Tiktok
-* Bluesky
-
-```{note} What websites the Auto Archiver can archie depends on what extractors you have enabled in
-your configuration. See [configuration](./configurations.md) for more info.
-```
-
-### Q: Does the Auto Archiver only work for social media posts ?
-**A:** No, the Auto Archiver can archive any web page on the internet, not just social media posts.
-However, for social media posts Auto Archiver can extract more relevant/useful information (such as 
-post comments, likes, author etc.) which may not be available for a generic website. If you are looking
-to more generally archive webpages, then you should make sure to enable the [](../modules/autogen/extractor/wacz_extractor_enricher.md)
-and the [](../modules/autogen/extractor/wayback_extractor_enricher.md).
-
-### Q: What kind of data is stored for each webpage that's archived?
-**A:** This depends on the website archived, but more generally, for social media posts any videos and photos in
-the post will be archived. For video sites, the video will be downloaded separately. For most of these sites, additional
-metadata such as published date, uploader/author and ratings/comments will also be saved. Additionally, further data can be
-saved depending on the enrichers that you have enabled. Some other types of data saved are timestamps if you have the 
-[](../modules/autogen/enricher/timestamping_enricher.md) or [](../modules/autogen/enricher/opentimestamps_enricher.md) enabled,
-screenshots of the web page with the [](../modules/autogen/enricher/screenshot_enricher.md), and for videos, thumbnails of the
-video with the [](../modules/autogen/enricher/thumbnail_enricher.md). You can also store things like hashes (SHA256, or pdq hashes)
-with the various hash enrichers.
-
-### Q: Where is my data stored?
-**A:** With the default configuration, data is stored on your local computer in the `local_storage` folder. You can adjust these settings by
-changing the [storage modules](../modules/storage.md) you have enabled. For example, you could choose to store your data in an S3 bucket or 
-on Google Drive. 
-
-```{note}
-You can choose to store your data in multiple places, for example your local drive **and** an S3 bucket for redundancy.
-```
-
-### Q: What should I do is something doesn't work?
-**A:** First, read through the log files to see if you can find a specific reason why something isn't working. Learn more about logging
-and how to enable debug logging in the [Logging Howto](../how_to/logging.md).
-
-If you cannot find an answer in the logs, then try searching this documentation or existing / closed issues on the [Github Issue Tracker](https://github.com/bellingcat/auto-archiver/issues?q=is%3Aissue%20). If you still cannot find an answer, then consider opening an issue on the Github Issue Tracker or asking in the Bellingcat Discord
-'Auto Archiver' group.
-
-#### Common reasons why an archiving might not work:
-
-* The website may have temporarily adjusted its settings - sometimes sites like Telegram or Twitter adjust their scraping protection settings. Often,
-waiting a day or two and then trying again can work.
-* The site requires you to be logged in - you could try using cookies or authentication to bypass any blocks. See [](../installation/authentication.md) for more information.
-* The website you're trying to archive has changed its settings/structure. Make sure you're using the latest version of Auto Archiver and try again.
--- a/docs/source/installation/installation.md
+++ b/docs/source/installation/installation.md
@@ -1,62 +0,0 @@
-# Installation
-
-```{toctree}
-:maxdepth: 1
-
-upgrading.md
-```
-
-There are 3 main ways to use the auto-archiver. We recommend the 'docker' method for most uses. This installs all the requirements in one command.
-
-1. Easiest (recommended): [via docker](#installing-with-docker)
-2. Local Install: [using pip](#installing-locally-with-pip)
-3. Developer Install: [see the developer guidelines](../development/developer_guidelines)
-
-## 1. Installing with Docker
-
-[![dockeri.co](https://dockerico.blankenship.io/image/bellingcat/auto-archiver)](https://hub.docker.com/r/bellingcat/auto-archiver)
-
-Docker works like a virtual machine running inside your computer, making installation simple. You'll need to first set up Docker, and then download the Auto Archiver 'image':
-
-
-**a) Download and install docker**
-
-Go to the [Docker website](https://docs.docker.com/get-docker/) and download right version for your operating system. 
-
-**b) Pull the Auto Archiver docker image**
-
-Open your command line terminal, and copy-paste / type:
-
-```bash
-docker pull bellingcat/auto-archiver
-```
-
-This will download the docker image, which may take a while.
-
-That's it, all done! You're now ready to set up [your configuration file](configurations.md). Or, if you want to use the recommended defaults, then you can [run Auto Archiver immediately](setup.md#running-a-docker-install).
-
------------
-
-## 2. Installing Locally with Pip
-
-1. Make sure you have python 3.10 or higher installed
-2. Install the package with your preferred package manager: `pip/pipenv/conda install auto-archiver` or `poetry add auto-archiver`
-3. Test it's installed with `auto-archiver --help`
-4. Install other local dependency requirements (for example `ffmpeg`, `firefox`)
-
-After this, you're ready to set up your [your configuration file](configurations.md), or if you want to use the recommended defaults, then you can [run Auto Archiver immediately](setup.md#running-a-local-install).
-
-### Installing Local Requirements
-
-If using the local installation method, you will also need to install the following dependencies locally:
-
-1.[ffmpeg](https://www.ffmpeg.org/) - for handling of downloaded videos
-2. [firefox](https://www.mozilla.org/en-US/firefox/new/) and [geckodriver](https://github.com/mozilla/geckodriver/releases) on a path folder like `/usr/local/bin` - for taking webpage screenshots with the screenshot enricher
-3. (optional) [fonts-noto](https://fonts.google.com/noto) to deal with multiple unicode characters during selenium/geckodriver's screenshots: `sudo apt install fonts-noto -y`.
-4. [Browsertrix Crawler docker image](https://hub.docker.com/r/webrecorder/browsertrix-crawler) for the WACZ enricher/archiver
-
-
-
-## Developer Install
-
-[See the developer guidelines](../development/developer_guidelines)
--- a/docs/source/installation/requirements.md
+++ b/docs/source/installation/requirements.md
@@ -1,14 +0,0 @@
-# Requirements
-
-Using the Auto Archiver is very simple, but ideally you have some familiarity with using the command line to run programs. ([Command line crash course](https://developer.mozilla.org/en-US/docs/Learn_web_development/Getting_started/Environment_setup/Command_line)).
-
-### System Requirements
-
-* Auto Archiver works on any Windows, macOS and Linux computer
-* If you're using the **local install** method, then you should make sure to have python3.10+ installed
-
-### Storage Requirements
-
-By default, Auto Archiver uses your local computer storage for any downloaded media (videos, images etc.). If you're downloading large files, this may take up a lot of your local computer's space (more than 5GB of space).
-
-If your storage space is limited, then you may want to set up an [alternative storage method](../modules/storage.md) for your media.
--- a/docs/source/installation/settings.html
+++ b/docs/source/installation/settings.html
--- a/docs/source/installation/setup.md
+++ b/docs/source/installation/setup.md
@@ -1,79 +0,0 @@
-# Getting Started
-
-```{toctree}
-:hidden:
-
-installation.md
-configurations.md
-config_editor.md
-authentication.md
-requirements.md
-faq.md
-config_cheatsheet.md
-```
-
-## Getting Started
-
-To get started with Auto Archiver, there are 3 main steps you need to complete.
-
-1. [Install Auto Archiver](installation.md)
-2. [Setup up your configuration](configurations.md) (if you are ok with the default settings, you can skip this step)
-3. Run the archiving process<a id="running"></a>
-
-The way you run the Auto Archiver depends on how you installed it (docker install or local install)
-
-### Running a Docker Install
-
-If you installed Auto Archiver using docker, open up your terminal, and copy-paste / type the following command:
-
-```bash
-docker run -it --rm -v $PWD/secrets:/app/secrets -v $PWD/local_archive:/app/local_archive bellingcat/auto-archiver
- ```
-
-breaking this command down:
-   1. `docker run` tells docker to start a new container (an instance of the image)
-   2. `-it` tells docker to run in 'interactive mode' so that we get nice colour logs
-   3. `--rm` makes sure this container is removed after execution (less garbage locally)
-   4. `-v $PWD/secrets:/app/secrets` - your secrets folder with settings
-      1. `-v` is a volume flag which means a folder that you have on your computer will be connected to a folder inside the docker container
-      2. `$PWD/secrets` points to a `secrets/` folder in your current working directory (where your console points to), we use this folder as a best practice to hold all the secrets/tokens/passwords/... you use
-      3. `/app/secrets` points to the path the docker container where this image can be found
-   5.  `-v $PWD/local_archive:/app/local_archive` - (optional) if you use local_storage
-       1.  `-v` same as above, this is a volume instruction
-       2.  `$PWD/local_archive` is a folder `local_archive/` in case you want to archive locally and have the files accessible outside docker
-       3.  `/app/local_archive` is a folder inside docker that you can reference in your orchestration.yml file 
-
-### Example invocations
-
-The invocations below will run the auto-archiver Docker image using a configuration file that you have specified
-
-```bash
-# Have auto-archiver run with the default settings, generating a settings file in ./secrets/orchestration.yaml
-docker run -it --rm -v $PWD/secrets:/app/secrets -v $PWD/local_archive:/app/local_archive bellingcat/auto-archiver
-
-# uses the same configuration, but with the `gsheet_feeder`, a header on row 2 and with some different column names
-# Note this expects you to have followed the [Google Sheets setup](how_to/google_sheets.md) and added your service_account.json to the `secrets/` folder
-# notice that columns is a dictionary so you need to pass it as JSON and it will override only the values provided
-docker run -it --rm -v $PWD/secrets:/app/secrets -v $PWD/local_archive:/app/local_archive bellingcat/auto-archiver --feeders=gsheet_feeder --gsheet_feeder.sheet="use it on another sheets doc" --gsheet_feeder.header=2 --gsheet_feeder.columns='{"url": "link"}'
-# Runs auto-archiver for the first time, but in 'full' mode, enabling all modules to get a full settings file
-docker run -it --rm -v $PWD/secrets:/app/secrets -v $PWD/local_archive:/app/local_archive bellingcat/auto-archiver --mode full
-```
-
------------
-
-### Running a Local Install
-
-### Example invocations
-
-Once all your [local requirements](#installing-local-requirements) are correctly installed, the
-
-```bash
-# all the configurations come from ./secrets/orchestration.yaml
-auto-archiver --config secrets/orchestration.yaml
-# uses the same configurations but for another google docs sheet 
-# with a header on row 2 and with some different column names
-# notice that columns is a dictionary so you need to pass it as JSON and it will override only the values provided
-auto-archiver --config secrets/orchestration.yaml --gsheet_feeder.sheet="use it on another sheets doc" --gsheet_feeder.header=2 --gsheet_feeder.columns='{"url": "link"}'
-# all the configurations come from orchestration.yaml and specifies that s3 files should be private
-auto-archiver --config secrets/orchestration.yaml --s3_storage.private=1
-```
--- a/docs/source/installation/upgrading.md
+++ b/docs/source/installation/upgrading.md
@@ -1,30 +0,0 @@
-
-# Upgrading
-
-If an update is available, then you will see a message in the logs when you
-run Auto Archiver. Here's what those logs look like:
-
-```{code} bash
-********* IMPORTANT: UPDATE AVAILABLE ********
-A new version of auto-archiver is available (v0.13.6, you have 0.13.4)
-Make sure to update to the latest version using: `pip install --upgrade auto-archiver`
-```
-
-Upgrading Auto Archiver depends on the way you installed it.
-
-## Docker
-
-To upgrade using docker, update the docker image with:
-
-```
-docker pull bellingcat/auto-archiver:latest 
-```
-
-## Pip
-
-To upgrade the pip package, use:
-
-```
-pip install --upgrade auto-archiver
-```
-
--- a/docs/source/modules/database.md
+++ b/docs/source/modules/database.md
@@ -1,15 +0,0 @@
-# Database Modules
-
-Database modules are used to store the status and results of the extraction and enrichment processes somewhere. The database modules are responsible for creating and managing entires for each item that has been processed.
-
-The default (enabled) databases are the CSV Database and the Console Database.
-
-```{include} autogen/database.md
-```
-
-```{toctree}
-:maxdepth: 1
-:hidden:
-:glob:
-autogen/database/*
-```
--- a/docs/source/modules/enricher.md
+++ b/docs/source/modules/enricher.md
@@ -1,14 +0,0 @@
-# Enricher Modules
-
-Enricher modules are used to add additional information to the items  that have been extracted. Common enrichment tasks include adding metadata to items, such as the hash of the item, a screenshot of the webpage when the item was extracted, or general metadata like the date and time the item was extracted.
-
-
-```{include} autogen/enricher.md
-```
-
-```{toctree}
-:maxdepth: 1
-:hidden:
-:glob:
-autogen/enricher/*
-```
--- a/docs/source/modules/extractor.md
+++ b/docs/source/modules/extractor.md
@@ -1,18 +0,0 @@
-# Extractor Modules
-
-Extractor modules are used to extract the content of a given URL. Typically, one extractor will work for one website or platform (e.g. a Telegram extractor or an Instagram), however, there are several wide-ranging extractors which work for a wide range of websites.
-
-Extractors that are able to extract content from a wide range of websites include:
-1. Generic Extractor: parses videos and images on sites using the powerful yt-dlp library.
-2. Wayback Machine Extractor: sends pages to the Wayback machine for archiving, and stores the link.
-3. WACZ Extractor: runs a web browser to 'browse' the URL and save a copy of the page in WACZ format. 
-
-```{include} autogen/extractor.md
-```
-
-```{toctree}
-:maxdepth: 1
-:hidden:
-:glob:
-autogen/extractor/*
-```
--- a/docs/source/modules/feeder.md
+++ b/docs/source/modules/feeder.md
@@ -1,20 +0,0 @@
-# Feeder Modules
-
-Feeder modules are used to feed URLs into the Auto Archiver for processing. Feeders can take these URLs from a variety of sources, such as a file, a database, or the command line.
-
-The default feeder is the command line feeder (`cli_feeder`), which allows you to input URLs directly into `auto-archiver` from the command line.
-
-Command line feeder usage:
-```{code} bash
-auto-archiver [options] -- URL1 URL2 ...
-```
-
-```{include} autogen/feeder.md
-```
-
-```{toctree}
-:maxdepth: 1
-:glob:
-:hidden:
-autogen/feeder/*
-```
--- a/docs/source/modules/formatter.md
+++ b/docs/source/modules/formatter.md
@@ -1,13 +0,0 @@
-# Formatter Modules
-
-Formatter modules are used to format the data extracted from a URL into a specific format. Currently the most widely-used formatter is the HTML formatter, which formats the data into an easily viewable HTML page.
-
-```{include} autogen/formatter.md
-```
-
-```{toctree}
-:maxdepth: 1
-:hidden:
-:glob:
-autogen/formatter/*
-```
--- a/docs/source/modules/storage.md
+++ b/docs/source/modules/storage.md
@@ -1,15 +0,0 @@
-# Storage Modules
-
-Storage modules are used to store the data extracted from a URL in a persistent location. This can be on your local hard disk, or on a remote server (e.g. S3 or Google Drive).
-
-The default is to store the files downloaded (e.g. images, videos) in a local directory.
-
-```{include} autogen/storage.md
-```
-
-```{toctree}
-:maxdepth: 1
-:hidden:
-:glob:
-autogen/storage/*
-```
--- a/docs/source/overview.md
+++ b/docs/source/overview.md
@@ -1,16 +0,0 @@
-
-```{include} ../../README.md
-```
-
-```{toctree}
-:maxdepth: 2
-:hidden:
-:caption: Contents:
-
-Overview <self>
-installation/installation.rst
-core_modules.md
-how_to
-development/developer_guidelines
-autoapi/index.rst
-```
--- a/example.orchestration.yaml
+++ b/example.orchestration.yaml
@@ -0,0 +1,125 @@
+steps:
+  # only 1 feeder allowed
+  feeder: gsheet_feeder # defaults to cli_feeder
+  archivers: # order matters, uncomment to activate
+    # - vk_archiver
+    # - telethon_archiver
+    # - telegram_archiver
+    # - twitter_archiver
+    # - twitter_api_archiver
+    # - instagram_tbot_archiver
+    # - instagram_archiver
+    # - tiktok_archiver
+    - youtubedl_archiver
+    # - wayback_archiver_enricher
+    # - wacz_archiver_enricher
+  enrichers:
+    - hash_enricher
+    # - metadata_enricher
+    # - screenshot_enricher
+    # - thumbnail_enricher
+    # - wayback_archiver_enricher
+    # - wacz_archiver_enricher
+    # - pdq_hash_enricher # if you want to calculate hashes for thumbnails, include this after thumbnail_enricher
+  formatter: html_formatter # defaults to mute_formatter
+  storages:
+    - local_storage
+    # - s3_storage
+    # - gdrive_storage
+  databases:
+    - console_db
+    # - csv_db
+    # - gsheet_db
+    # - mongo_db
+
+configurations:
+  gsheet_feeder:
+    sheet: "your sheet name"
+    header: 1
+    service_account: "secrets/service_account.json"
+    # allow_worksheets: "only parse this worksheet"
+    # block_worksheets: "blocked sheet 1,blocked sheet 2"
+    use_sheet_names_in_stored_paths: false
+    columns:
+      url: link
+      status: archive status
+      folder: destination folder
+      archive: archive location
+      date: archive date
+      thumbnail: thumbnail
+      timestamp: upload timestamp
+      title: upload title
+      text: textual content
+      screenshot: screenshot
+      hash: hash
+      pdq_hash: perceptual hashes
+      wacz: wacz
+      replaywebpage: replaywebpage
+  instagram_tbot_archiver:
+    api_id: "TELEGRAM_BOT_API_ID"
+    api_hash: "TELEGRAM_BOT_API_HASH"
+    # session_file: "secrets/anon"
+  telethon_archiver:
+    api_id: "TELEGRAM_BOT_API_ID"
+    api_hash: "TELEGRAM_BOT_API_HASH"
+    # session_file: "secrets/anon"
+    join_channels: false
+    channel_invites: # if you want to archive from private channels
+      - invite: https://t.me/+123456789
+        id: 0000000001
+      - invite: https://t.me/+123456788
+        id: 0000000002
+
+  twitter_api_archiver:
+    # either bearer_token only
+    bearer_token: "TWITTER_BEARER_TOKEN"
+    # OR all of the below
+    # consumer_key: ""
+    # consumer_secret: ""
+    # access_token: ""
+    # access_secret: ""
+  instagram_archiver:
+    username: "INSTAGRAM_USERNAME"
+    password: "INSTAGRAM_PASSWORD"
+    # session_file: "secrets/instaloader.session"
+
+  vk_archiver:
+    username: "or phone number"
+    password: "vk pass"
+    session_file: "secrets/vk_config.v2.json"
+
+  screenshot_enricher:
+    width: 1280
+    height: 2300
+  wayback_archiver_enricher:
+    timeout: 10
+    key: "wayback key"
+    secret: "wayback secret"
+  hash_enricher:
+    algorithm: "SHA3-512" # can also be SHA-256
+  wacz_archiver_enricher:
+    profile: secrets/profile.tar.gz
+  local_storage:
+    save_to: "./local_archive"
+    save_absolute: true
+    filename_generator: static
+    path_generator: flat
+  s3_storage:
+    bucket: your-bucket-name
+    region: reg1
+    key: S3_KEY
+    secret: S3_SECRET
+    endpoint_url: "https://{region}.digitaloceanspaces.com"
+    cdn_url: "https://{bucket}.{region}.cdn.digitaloceanspaces.com/{key}"
+    # if private:true S3 urls will not be readable online
+    private: false
+    # with 'random' you can generate a random UUID for the URL instead of a predictable path, useful to still have public but unlisted files, alternative is 'default' or not omitted from config
+    key_path: random
+  gdrive_storage:
+    path_generator: url
+    filename_generator: random
+    root_folder_id: folder_id_from_url
+    oauth_token: secrets/gd-token.json # needs to be generated with scripts/create_update_gdrive_oauth_token.py
+    service_account: "secrets/service_account.json"
+  csv_db:
+    csv_file: "./local_archive/db.csv"
--- a/poetry.lock
+++ b/poetry.lock
--- a/pyproject.toml
+++ b/pyproject.toml
@@ -1,120 +1,4 @@
 [build-system]
-requires = ["poetry-core>=2.0.0,<3.0.0"]
-build-backend = "poetry.core.masonry.api"
-
-[project]
-name = "auto-archiver"
-version = "0.13.6"
-description = "Automatically archive links to videos, images, and social media content from Google Sheets (and more)."
-
-requires-python = ">=3.10,<3.13"
-license = "MIT"
-authors = [
-    { name = "Bellingcat", email = "tech@bellingcat.com" },
-]
-readme = "README.md"
-keywords = ["archive", "oosi", "osint", "scraping"]
-classifiers = [
-    "Intended Audience :: Developers",
-    "Intended Audience :: Science/Research",
-    "License :: OSI Approved :: MIT License",
-    "Programming Language :: Python :: 3"
-]
-
-dependencies = [
-    "gspread (>=0.0.0)",
-    "beautifulsoup4 (>=0.0.0)",
-    "bs4 (>=0.0.0)",
-    "loguru (>=0.0.0)",
-    "ffmpeg-python (>=0.0.0)",
-    "selenium (>=0.0.0)",
-    "telethon (>=0.0.0)",
-    "google-api-python-client (>=0.0.0)",
-    "google-auth-httplib2 (>=0.0.0)",
-    "google-auth-oauthlib (>=0.0.0)",
-    "oauth2client (>=0.0.0)",
-    "pdqhash (>=0.0.0)",
-    "pillow (>=0.0.0)",
-    "python-slugify (>=0.0.0)",
-    "dateparser (>=0.0.0)",
-    "python-twitter-v2 (>=0.0.0)",
-    "instaloader (>=0.0.0)",
-    "tqdm (>=0.0.0)",
-    "jinja2 (>=0.0.0)",
-    "pyOpenSSL (==24.2.1)",
-    "cryptography (>=41.0.0,<42.0.0)",
-    "boto3 (>=1.28.0,<2.0.0)",
-    "dataclasses-json (>=0.0.0)",
-    "yt-dlp (>=2025.1.26,<2026.0.0)",
-    "numpy (==2.1.3)",
-    "vk-url-scraper (>=0.0.0)",
-    "requests[socks] (>=0.0.0)",
-    "warcio (>=0.0.0)",
-    "jsonlines (>=0.0.0)",
-    "pysubs2 (>=0.0.0)",
-    "retrying (>=0.0.0)",
-    "tsp-client (>=0.0.0)",
-    "certvalidator (>=0.0.0)",
-    "rich-argparse (>=1.6.0,<2.0.0)",
-    "ruamel-yaml (>=0.18.10,<0.19.0)",
-    "opentimestamps (>=0.4.5,<0.5.0)",
-]
-
-[tool.poetry.group.dev.dependencies]
-pytest = "^8.3.4"
-autopep8 = "^2.3.1"
-pytest-loguru = "^0.4.0"
-pytest-mock = "^3.14.0"
-ruff = "^0.9.10"
-pre-commit = "^4.1.0"
-
-[tool.poetry.group.docs.dependencies]
-sphinx = "^8.1.3"
-sphinx-autoapi = "^3.4.0"
-sphinxcontrib-mermaid = "^1.0.0"
-sphinx-autobuild = "^2024.10.3"
-sphinx-copybutton = "^0.5.2"
-myst-parser = "^4.0.0"
-sphinx-book-theme = "^1.1.3"
-linkify-it-py = "^2.0.3"
-
-
-[project.scripts]
-auto-archiver = "auto_archiver.__main__:main"
-
-[project.urls]
-homepage = "https://github.com/bellingcat/auto-archiver"
-repository = "https://github.com/bellingcat/auto-archiver"
-documentation = "https://github.com/bellingcat/auto-archiver"
-
-
-[tool.pytest.ini_options]
-markers = [
-    "download: marks tests that download content from the network",
-    "incremental: marks a class to run tests incrementally. If a test fails in the class, the remaining tests will be skipped",
-]
-
-[tool.ruff]
-#exclude = ["docs"]
-line-length = 120
-# Remove this for a more detailed lint report
-output-format = "concise"
-# TODO: temp ignore rule for timestamping_enricher to allow for open PR
-exclude = ["src/auto_archiver/modules/timestamping_enricher/*"]
-
-
-[tool.ruff.lint]
-# Extend the rules to check for by adding them to this option:
-# See documentation for more details: https://docs.astral.sh/ruff/rules/
-#extend-select = ["B"]
-
-[tool.ruff.lint.per-file-ignores]
-# Ignore import violations in __init__.py files
-"__init__.py" = ["F401", "F403"]
-# Ignore 'useless expression' in manifest files.
-"__manifest__.py" = ["B018"]
-
-
-[tool.ruff.format]
-docstring-code-format = false
-
+requires = ["setuptools", "wheel", "setuptools-pipfile"]
+build-backend = "setuptools.build_meta"
+[tool.setuptools-pipfile]
--- a/scripts/create_update_gdrive_oauth_token.py
+++ b/scripts/create_update_gdrive_oauth_token.py
@@ -1,6 +1,5 @@
 import os.path
-import click
-import json
+import click, json

 from google.auth.transport.requests import Request
 from google.oauth2.credentials import Credentials
@@ -13,7 +12,7 @@ from googleapiclient.errors import HttpError
 # Code below from https://developers.google.com/drive/api/quickstart/python
 # Example invocation: py scripts/create_update_gdrive_oauth_token.py -c secrets/credentials.json -t secrets/gd-token.json

-SCOPES = ["https://www.googleapis.com/auth/drive.file"]
+SCOPES = ['https://www.googleapis.com/auth/drive']


@click.command(
@@ -24,7 +23,7 @@ SCOPES = ["https://www.googleapis.com/auth/drive.file"]
    "-c",
    type=click.Path(exists=True),
    help="path to the credentials.json file downloaded from https://console.cloud.google.com/apis/credentials",
-    required=True,
+    required=True
 )
@click.option(
    "--token",
@@ -32,58 +31,59 @@ SCOPES = ["https://www.googleapis.com/auth/drive.file"]
    type=click.Path(exists=False),
    default="gd-token.json",
    help="file where to place the OAuth token, defaults to gd-token.json which you must then move to where your orchestration file points to, defaults to gd-token.json",
-    required=True,
+    required=True
 )
 def main(credentials, token):
    # The file token.json stores the user's access and refresh tokens, and is
    # created automatically when the authorization flow completes for the first time.
    creds = None
    if os.path.exists(token):
-        with open(token, "r") as stream:
+        with open(token, 'r') as stream:
            creds_json = json.load(stream)
            # creds = Credentials.from_authorized_user_file(creds_json, SCOPES)
-            creds_json["refresh_token"] = creds_json.get("refresh_token", "")
+            creds_json['refresh_token'] = creds_json.get("refresh_token", "")
            creds = Credentials.from_authorized_user_info(creds_json, SCOPES)

    # If there are no (valid) credentials available, let the user log in.
    if not creds or not creds.valid:
        if creds and creds.expired and creds.refresh_token:
-            print("Requesting new token")
+            print('Requesting new token')
            creds.refresh(Request())
        else:
-            print("First run through so putting up login dialog")
+            print('First run through so putting up login dialog')
            # credentials.json downloaded from https://console.cloud.google.com/apis/credentials
            flow = InstalledAppFlow.from_client_secrets_file(credentials, SCOPES)
            creds = flow.run_local_server(port=55192)
        # Save the credentials for the next run
-        with open(token, "w") as token:
-            print("Saving new token")
+        with open(token, 'w') as token:
+            print('Saving new token')
            token.write(creds.to_json())
    else:
-        print("Token valid")
+        print('Token valid')

    try:
-        service = build("drive", "v3", credentials=creds)
+        service = build('drive', 'v3', credentials=creds)

        # About the user
        results = service.about().get(fields="*").execute()
-        emailAddress = results["user"]["emailAddress"]
+        emailAddress = results['user']['emailAddress']
        print(emailAddress)

        # Call the Drive v3 API and return some files
-        results = service.files().list(pageSize=10, fields="nextPageToken, files(id, name)").execute()
-        items = results.get("files", [])
+        results = service.files().list(
+            pageSize=10, fields="nextPageToken, files(id, name)").execute()
+        items = results.get('files', [])

        if not items:
-            print("No files found.")
+            print('No files found.')
            return
-        print("Files:")
+        print('Files:')
        for item in items:
-            print("{0} ({1})".format(item["name"], item["id"]))
+            print(u'{0} ({1})'.format(item['name'], item['id']))

    except HttpError as error:
-        print(f"An error occurred: {error}")
+        print(f'An error occurred: {error}')


-if __name__ == "__main__":
+if __name__ == '__main__':
    main()
--- a/scripts/generate_settings_schema.py
+++ b/scripts/generate_settings_schema.py
@@ -1,63 +0,0 @@
-import json
-import os
-import io
-
-from ruamel.yaml import YAML
-
-from auto_archiver.core.module import ModuleFactory
-from auto_archiver.core.consts import MODULE_TYPES
-from auto_archiver.core.config import EMPTY_CONFIG
-
-
-class SchemaEncoder(json.JSONEncoder):
-    def default(self, obj):
-        if isinstance(obj, set):
-            return list(obj)
-        return json.JSONEncoder.default(self, obj)
-
-
-# Get available modules
-module_factory = ModuleFactory()
-available_modules = module_factory.available_modules()
-
-modules_by_type = {}
-# Categorize modules by type
-for module in available_modules:
-    for type in module.manifest.get("type", []):
-        modules_by_type.setdefault(type, []).append(module)
-
-all_modules_ordered_by_type = sorted(
-    available_modules, key=lambda x: (MODULE_TYPES.index(x.type[0]), not x.requires_setup)
-)
-
-yaml: YAML = YAML()
-
-config_string = io.BytesIO()
-yaml.dump(EMPTY_CONFIG, config_string)
-config_string = config_string.getvalue().decode("utf-8")
-output_schema = {
-    "modules": dict(
-        (
-            module.name,
-            {
-                "name": module.name,
-                "display_name": module.display_name,
-                "manifest": module.manifest,
-                "configs": module.configs or None,
-            },
-        )
-        for module in all_modules_ordered_by_type
-    ),
-    "steps": dict(
-        (f"{module_type}s", [module.name for module in modules_by_type[module_type]]) for module_type in MODULE_TYPES
-    ),
-    "configs": [m.name for m in all_modules_ordered_by_type if m.configs],
-    "module_types": MODULE_TYPES,
-    "empty_config": config_string,
-}
-
-current_file_dir = os.path.dirname(os.path.abspath(__file__))
-output_file = os.path.join(current_file_dir, "settings/src/schema.json")
-with open(output_file, "w") as file:
-    print(f"Writing schema to {output_file}")
-    json.dump(output_schema, file, indent=4, cls=SchemaEncoder)
--- a/scripts/release.sh
+++ b/scripts/release.sh
@@ -0,0 +1,19 @@
+
+#!/bin/bash
+
+set -e
+
+TAG=$(python -c 'from src.auto_archiver.version import __version__; print("v" + __version__)')
+
+read -p "Creating new release for $TAG. Do you want to continue? [Y/n] " prompt
+
+if [[ $prompt == "y" || $prompt == "Y" || $prompt == "yes" || $prompt == "Yes" ]]; then
+    # git add -A
+    # git commit -m "Bump version to $TAG for release" || true && git push
+    echo "Creating new git tag $TAG"
+    git tag "$TAG" -m "$TAG"
+    git push --tags
+else
+    echo "Cancelled"
+    exit 1
+fi
--- a/scripts/settings/.gitignore
+++ b/scripts/settings/.gitignore
@@ -1,24 +0,0 @@
-# Logs
-logs
-*.log
-npm-debug.log*
-yarn-debug.log*
-yarn-error.log*
-pnpm-debug.log*
-lerna-debug.log*
-
-node_modules
-dist
-dist-ssr
-*.local
-
-# Editor directories and files
-.vscode/*
-!.vscode/extensions.json
-.idea
-.DS_Store
-*.suo
-*.ntvs*
-*.njsproj
-*.sln
-*.sw?
--- a/scripts/settings/index.html
+++ b/scripts/settings/index.html
@@ -1,3 +0,0 @@
-
-    <div id="root"></div>
-    <script type="module" src="/src/main.tsx"></script>
--- a/scripts/settings/package-lock.json
+++ b/scripts/settings/package-lock.json
--- a/scripts/settings/package.json
+++ b/scripts/settings/package.json
@@ -1,31 +0,0 @@
-{
-  "name": "material-ui-vite-ts",
-  "private": true,
-  "version": "5.0.0",
-  "type": "module",
-  "scripts": {
-    "dev": "vite",
-    "build": "vite build",
-    "preview": "vite preview"
-  },
-  "dependencies": {
-    "@dnd-kit/core": "^6.3.1",
-    "@dnd-kit/sortable": "^10.0.0",
-    "@emotion/react": "latest",
-    "@emotion/styled": "latest",
-    "@mui/icons-material": "^6.4.7",
-    "@mui/material": "latest",
-    "react": "19.0.0",
-    "react-dom": "19.0.0",
-    "react-markdown": "^10.0.0",
-    "yaml": "^2.7.0"
-  },
-  "devDependencies": {
-    "@types/react": "latest",
-    "@types/react-dom": "latest",
-    "@vitejs/plugin-react": "latest",
-    "typescript": "latest",
-    "vite": "latest",
-    "vite-plugin-singlefile": "^2.1.0"
-  }
-}
--- a/scripts/settings/src/App.tsx
+++ b/scripts/settings/src/App.tsx
@@ -1,450 +0,0 @@
-import * as React from 'react';
-import { useEffect, useState, useRef } from 'react';
-import Container from '@mui/material/Container';
-import Typography from '@mui/material/Typography';
-import Box from '@mui/material/Box';
-import FileUploadIcon from '@mui/icons-material/FileUpload';
-
-import {
-  DndContext,
-  closestCenter,
-  KeyboardSensor,
-  PointerSensor,
-  useSensor,
-  useSensors,
-  DragOverlay
-} from "@dnd-kit/core";
-import {
-  arrayMove,
-  SortableContext,
-  sortableKeyboardCoordinates,
-  rectSortingStrategy
-} from "@dnd-kit/sortable";
-
-import type { DragStartEvent, DragEndEvent, UniqueIdentifier } from "@dnd-kit/core";
-
-
-import { Module } from './types';
-
-import { modules, steps, module_types, empty_config } from './schema.json';
-import {
-  Stack,
-  Button,
-} from '@mui/material';
-import Grid from '@mui/material/Grid2';
-
-import { parseDocument, Document, YAMLSeq, YAMLMap, Scalar } from 'yaml'
-import StepCard from './StepCard';
-
-
-function FileDrop({ setYamlFile }: { setYamlFile: React.Dispatch<React.SetStateAction<Document>> }) {
-
-  const [showError, setShowError] = useState(false);
-  const [label, setLabel] = useState(<>Drag and drop your orchestration.yaml file here, or click to select a file.</>);
-  const wrapperRef = useRef(null);
-
-  function openYAMLFile(event: any) {
-    let file = event.target.files[0];
-    if (file.type.indexOf('yaml') === -1) {
-      setShowError(true);
-      setLabel(<>Invalid type, only YAML files are accepted.</>)
-      return;
-    }
-    let reader = new FileReader();
-    reader.onload = function (e) {
-      let contents = e.target ? e.target.result : '';
-      try {
-        let document = parseDocument(contents as string);
-        if (document.errors.length > 0) {
-          // not a valid yaml file
-          setShowError(true);
-          setLabel(<>Invalid file. Make sure your Orchestration is a valid YAML file with a 'steps' section in it.</>)
-          return;
-        } else {
-          setShowError(false);
-          setLabel(<>File loaded successfully.</>)
-        }
-        // do some basic validation of 'steps'
-        let steps = document.get('steps');
-        if (!steps) {
-          setShowError(true);
-          setLabel(<>Invalid file. Your orchestration file must have a 'steps' section in it.</>)
-          return;
-        }
-        const replacements = {
-          feeder: 'feeders',
-          formatter: 'formatters',
-          archivers: 'extractors',
-        };
-
-        let error = false;
-        for (let stepType of Object.keys(replacements)) {
-          if (steps.get(stepType) !== undefined) {
-            setShowError(true);
-            setLabel(<>Invalid file. Your orchestration file appears to be in the old (v0.12) format with a '{stepType}' section.<br/>You should manually update your orchestration file first (hint: {stepType} → {replacements[stepType]})</>);
-            error = true;
-            return;
-          }
-        };
-        setYamlFile(document);
-      } catch (e) {
-        console.error(e);
-      }
-    }
-    reader.readAsText(file);
-  }
-  return (
-    <>
-      <div
-      style={{
-        position: 'relative',
-        width: '100%',
-        border: 'dashed',
-        borderRadius:'5px',
-        textAlign: 'center',
-        borderWidth: '1px',
-        padding: '20px' }}
-      onDragEnter={(e) => {
-        e.currentTarget.style.backgroundColor = 'var(--mui-palette-LinearProgress-infoBg)';
-      }}
-      onDragLeave={(e) => {
-        e.currentTarget.style.backgroundColor = '';
-      }}
-      onDrop={(e) => {
-        e.currentTarget.style.backgroundColor = '';
-      }}
-      >
-        <FileUploadIcon style={{ fontSize: 50 }} />
-        <input style={{
-          opacity: 0,
-          position: 'absolute',
-          top: 0,
-          left: 0,
-          width: '100%',
-          height: '100%',
-          cursor: 'pointer',
-        }}
-        type="file" id="file"
-        accept=".yaml"
-        onChange={openYAMLFile} />
-        <Typography variant="body1" color={showError ? 'error' : ''} >
-          {label}
-        </Typography>
-      </div>
-    </>
-  );
-}
-
-function ModuleTypes({ stepType, setEnabledModules, enabledModules, configValues }: { stepType: string, setEnabledModules: any, enabledModules: any, configValues: any }) {
-  const [showError, setShowError] = useState<boolean>(false);
-  const [activeId, setActiveId] = useState<UniqueIdentifier>();
-  const [items, setItems] = useState<string[]>([]);
-
-  useEffect(() => {
-    setItems(enabledModules[stepType].map(([name, enabled]: [string, boolean]) => name));
-  }
-    , [enabledModules]);
-
-  const toggleModule = (event: any) => {
-    // make sure that 'feeder' and 'formatter' types only have one value
-    let name = event.target.id;
-    let checked = event.target.checked;
-    if (stepType === 'feeders' || stepType === 'formatters') {
-      // check how many modules of this type are enabled
-      const checkedModules = enabledModules[stepType].filter(([m, enabled]: [string, boolean]) => {
-        return (m !== name && enabled) || (checked && m === name)
-      });
-      if (checkedModules.length > 1) {
-        setShowError(true);
-      } else {
-        setShowError(false);
-      }
-    } else {
-      setShowError(false);
-    }
-    let newEnabledModules = { ...enabledModules };
-    newEnabledModules[stepType] = enabledModules[stepType].map(([m, enabled]: [string, boolean]) => {
-        return (m === name) ? [m, checked] : [m, enabled];
-      });
-    setEnabledModules(newEnabledModules);
-  }
-
-  const sensors = useSensors(
-    useSensor(PointerSensor),
-    useSensor(KeyboardSensor, {
-      coordinateGetter: sortableKeyboardCoordinates
-    })
-  );
-
-  const handleDragStart = (event: DragStartEvent) => {
-    setActiveId(event.active.id);
-  };
-
-  const handleDragEnd = (event: DragEndEvent) => {
-    setActiveId(undefined);
-    const { active, over } = event;
-
-    if (active.id !== over?.id) {
-      const oldIndex = items.indexOf(active.id as string);
-      const newIndex = items.indexOf(over?.id as string);
-
-      let newArray = arrayMove(items, oldIndex, newIndex);
-      // set it also on steps
-      let newEnabledModules = { ...enabledModules };
-      newEnabledModules[stepType] = enabledModules[stepType].sort((a, b) => {
-        return newArray.indexOf(a[0]) - newArray.indexOf(b[0]);
-      })
-      setEnabledModules(newEnabledModules);
-    }
-  };
-  return (
-    <>
-      <Box sx={{ my: 4 }}>
-        <Typography id={stepType} variant="h6" style={{ textTransform: 'capitalize' }} >
-          {stepType}
-        </Typography>
-        <Typography variant="body1" >
-          Select the <a href={`https://auto-archiver.readthedocs.io/en/latest/modules/${stepType.slice(0,-1)}.html`} target="_blank">{stepType}</a> you wish to enable. Drag to reorder.
-        </Typography>
-      </Box>
-      {showError ? <Typography variant="body1" color="error" >Only one {stepType.slice(0,-1)} can be enabled at a time.</Typography> : null}
-
-      <DndContext
-        sensors={sensors}
-        collisionDetection={closestCenter}
-        onDragEnd={handleDragEnd}
-        onDragStart={handleDragStart}
-      >
-        <Grid container spacing={1} key={stepType}>
-          <SortableContext items={items} strategy={rectSortingStrategy}>
-            {items.map((name: string) => {
-              let m: Module = modules[name];
-              return (
-                <StepCard key={name} type={stepType} module={m} toggleModule={toggleModule} enabledModules={enabledModules} configValues={configValues} />
-              );
-            })}
-            <DragOverlay>
-              {activeId ? (
-                <div
-                  style={{
-                    width: "100%",
-                    height: "100%",
-                    backgroundColor: "grey",
-                    opacity: 0.1,
-                  }}
-                ></div>
-
-              ) : null}
-            </DragOverlay>
-          </SortableContext>
-        </Grid>
-      </DndContext>
-    </>
-  );
-}
-
-
-export default function App() {
-  const [yamlFile, setYamlFile] = useState<Document>(new Document());
-  const [enabledModules, setEnabledModules] = useState<{}>(Object.fromEntries(Object.keys(steps).map(type => [type, steps[type].map((name: string) => [name, false])])));
-  const [configValues, setConfigValues] = useState<{
-    [key: string]: {
-      [key: string
-      ]: any
-    }
-  }>(
-    Object.keys(modules).reduce((acc, module) => {
-      acc[module] = {};
-      return acc;
-    }, {})
-  );
-
-  const saveSettings = function (copy: boolean = false) {
-    // edit the yamlFile
-
-    // generate the steps config
-    let stepsConfig = enabledModules;
-
-    let finalYamlFile: Document = null;
-    if (!yamlFile || yamlFile.contents == null) {
-      // create the yaml file from 
-      finalYamlFile = parseDocument(empty_config as string);
-    } else {
-      finalYamlFile = yamlFile;
-    }
-
-    // set the steps
-    module_types.forEach((type: string) => {
-      let stepType = type + 's';
-      let existingSteps  = finalYamlFile.getIn(['steps', stepType]) as YAMLSeq;
-      stepsConfig[stepType].forEach(([name, enabled]: [string, boolean]) => {
-        let index = existingSteps.items.findIndex((item) => { 
-          return (item.value || item) === name
-        });
-        let stepItem = finalYamlFile.getIn(['steps', stepType], true) as YAMLSeq;
-
-        if (enabled && index === -1) {
-            finalYamlFile.addIn(['steps', stepType], name);
-            stepItem.commentBefore = stepItem.commentBefore?.replace("\n - " + name, '');
-            stepItem.comment = stepItem.comment?.replace("\n - " + name, '');
-        } else if (!enabled && index !== -1) {
-            // set the value to empty and add a comment before with the commented value
-            finalYamlFile.deleteIn(['steps', stepType, index]);
-            stepItem.commentBefore += "\n - " + name;
-            finalYamlFile.setIn(['steps', stepType], stepItem);
-        }
-      });
-      // sort the items
-      existingSteps.items.sort((a: Scalar | string, b: Scalar | string) => {
-        return (stepsConfig[stepType].findIndex((val: [string, boolean]) => {return val[0] === (a.value || a)}) -
-                stepsConfig[stepType].findIndex((val: [string, boolean]) => {return val[0] === (b.value || b)}))
-      });
-      existingSteps.flow = existingSteps.items.length ? false : true;
-    });
-
-    // set all other settings
-    // loop through each item that isn't 'steps' in the finalYamlFile and check if it exists in configValues
-
-        Object.keys(configValues).forEach((module_name: string) => {
-          // get an existing key
-          let existingConfig = finalYamlFile.get(module_name, true) as YAMLMap;
-          if (existingConfig) {
-            Object.keys(configValues[module_name]).forEach((config_name: string) => {
-              let existingConfigYAML = existingConfig.get(config_name, true) as Scalar;
-              if (existingConfigYAML) {
-                existingConfigYAML.value = configValues[module_name][config_name];
-                existingConfig.set(config_name, existingConfigYAML);
-              } else {
-                existingConfig.set(config_name, configValues[module_name][config_name]);
-              }
-            });
-            finalYamlFile.set(module_name, existingConfig);
-          } else {
-            if (configValues[module_name] && Object.keys(configValues[module_name]).length > 0) {
-              finalYamlFile.set(module_name, configValues[module_name]);
-            }
-          }
-        });
-
-    if (copy) {
-      navigator.clipboard.writeText(String(finalYamlFile)).then(() => {
-        alert("Settings copied to clipboard.");
-      });
-    } else {
-      // offer the file for download
-      const blob = new Blob([String(finalYamlFile)], { type: 'application/x-yaml' });
-      const url = URL.createObjectURL(blob);
-      const a = document.createElement('a');
-      a.href = url;
-      a.download = 'orchestration.yaml';
-      a.click();
-    }
-  }
-
-  useEffect(() => {
-    // load the configs, and set the default values if they exist
-    let newConfigValues = {};
-    Object.keys(modules).map((module: string) => {
-      let m = modules[module];
-      let configs = m.configs;
-      if (!configs) {
-        return;
-      }
-      newConfigValues[module] = {};
-      Object.keys(configs).map((config: string) => {
-        let config_args = configs[config];
-        if (config_args.default !== undefined) {
-          newConfigValues[module][config] = config_args.default;
-        }
-      });
-    })
-    setConfigValues(newConfigValues);
-  }, []);
-
-  useEffect(() => {
-    if (!yamlFile || yamlFile.contents == null) {
-      return;
-    }
-
-    let settings = yamlFile.toJS();
-    // make a deep copy of settings
-    let stepSettings = settings['steps'];
-   
-    let newEnabledModules = Object.fromEntries(Object.keys(steps).map((type: string) => {
-      return [type, steps[type].map((name: string) => {
-        return [name, stepSettings[type].indexOf(name) !== -1];
-      }).sort((a, b) => {
-        let aIndex = stepSettings[type].indexOf(a[0]);
-        let bIndex = stepSettings[type].indexOf(b[0]);
-        if (aIndex === -1 && bIndex === -1) {
-          return a - b;
-        }
-        if (bIndex === -1) {
-          return -1;
-        }
-        if (aIndex === -1) {
-          return 1;
-        }
-        return aIndex - bIndex;
-      })];
-    }).sort((a, b) => {
-      return module_types.indexOf(a[0]) - module_types.indexOf(b[0]);
-    }));
-    setEnabledModules(newEnabledModules);
-
-    // set the config values
-    let newConfigValues = settings;
-    delete newConfigValues['steps'];
-
-
-    setConfigValues(Object.keys(modules).reduce((acc, module) => {
-      acc[module] = newConfigValues[module] || {};
-      return acc;
-    }, {}));
-  }, [yamlFile]);
-
-
-
-  return (
-    <Container maxWidth="lg">
-      <Box sx={{ my: 4 }}>
-        <Box sx={{ my: 4 }}>
-          <Typography variant="h5" >
-            1. Select your orchestration.yaml settings file.
-          </Typography>
-          <Typography variant="body1">Or skip this step to start from scratch</Typography>
-          <FileDrop setYamlFile={setYamlFile} />
-        </Box>
-        <Box sx={{ my: 4 }}>
-          <Typography variant="h5" >
-            2. Choose the Modules you wish to enable/disable
-          </Typography>
-          {Object.keys(steps).map((stepType: string) => {
-            return (
-              <Box key={stepType} sx={{ my: 4 }}>
-                <ModuleTypes stepType={stepType} setEnabledModules={setEnabledModules} enabledModules={enabledModules} configValues={configValues} />
-              </Box>
-            );
-          })}
-        </Box>
-        <Box sx={{ my: 4 }}>
-          <Typography variant="h5" >
-            3. Configure your Enabled Modules
-          </Typography>
-          <Typography variant="body1" >
-            Next to each module you've enabled, you can click 'Configure' to set the module's settings.
-          </Typography>
-        </Box>
-        <Box sx={{ my: 4 }}>
-          <Typography variant="h5" >
-            4. Save your settings
-          </Typography>
-          <Stack direction="row" spacing={2} sx={{ my: 2 }}>
-            <Button variant="contained" color="primary" onClick={() => saveSettings(true)}>Copy Settings to Clipboard</Button>
-            <Button variant="contained" color="primary" onClick={() => saveSettings()}>Save Settings to File</Button>
-          </Stack>
-        </Box>
-      </Box>
-    </Container>
-  );
-}
--- a/scripts/settings/src/StepCard.tsx
+++ b/scripts/settings/src/StepCard.tsx
@@ -1,258 +0,0 @@
-import { useState } from "react";
-import { useSortable } from "@dnd-kit/sortable";
-import ReactMarkdown from 'react-markdown';
-
-import { CSS } from "@dnd-kit/utilities";
-
-import {
-    Card,
-    CardActions,
-    CardHeader,
-    Button,
-    Dialog,
-    DialogTitle,
-    DialogContent,
-    Box,
-    IconButton,
-    Checkbox,
-    Select,
-    MenuItem,
-    FormControl,
-    FormControlLabel,
-    FormHelperText,
-    TextField,
-    Stack,
-    Typography,
-    InputAdornment,
-} from '@mui/material';
-import Grid from '@mui/material/Grid2';
-import DragIndicatorIcon from '@mui/icons-material/DragIndicator';
-import Visibility from '@mui/icons-material/Visibility';
-import VisibilityOff from '@mui/icons-material/VisibilityOff';
-import HelpIconOutlined from '@mui/icons-material/HelpOutline';
-import { Module, Config } from "./types";
-
-
-// adds 'capitalize' method to String prototype
-declare global {
-    interface String {
-        capitalize(): string;
-    }
-}
-String.prototype.capitalize = function (this: string) {
-    return this.charAt(0).toUpperCase() + this.slice(1);
-};
-
-const StepCard = ({
-    type,
-    module,
-    toggleModule,
-    enabledModules,
-    configValues
-}: {
-    type: string,
-    module: Module,
-    toggleModule: any,
-    enabledModules: any,
-    configValues: any
-}) => {
-    const {
-        attributes,
-        listeners,
-        setNodeRef,
-        transform,
-        transition,
-        isDragging
-    } = useSortable({ id: module.name });
-
-
-    const style = {
-        ...Card.style,
-        transform: CSS.Transform.toString(transform),
-        transition,
-        zIndex: isDragging ? "100" : "auto",
-        opacity: isDragging ? 0.3 : 1
-    };
-
-    let name = module.name;
-    const [helpOpen, setHelpOpen] = useState(false);
-    const [configOpen, setConfigOpen] = useState(false);
-    const enabled = enabledModules[type].find((m: any) => m[0] === name)[1];
-
-    return (
-        <Grid ref={setNodeRef} size={{ xs: 6, sm: 4, md: 3 }} style={style}>
-            <Card >
-                <CardHeader
-                    title={
-                        <FormControlLabel
-                            style={{paddingRight: '0 !important'}}
-                            control={<Checkbox title="Check to enable this module" sx={{paddingTop:0, paddingBottom:0}} id={name} onClick={toggleModule} checked={enabled} />}
-                            label={module.display_name} />
-                    }
-                />
-                <CardActions>
-                    <Box sx={{ justifyContent: 'space-between', display: 'flex', width: '100%' }}>
-                        <Box>
-                    <IconButton title="Module information" size="small" onClick={() => setHelpOpen(true)}>
-                        <HelpIconOutlined />
-                    </IconButton>
-                    {enabled && module.configs && name != 'cli_feeder' ? (
-                        <Button size="small" onClick={() => setConfigOpen(true)}>Configure</Button>
-                    ) : null}
-                    </Box>
-                    <IconButton size="small" title="Drag to reorder" sx={{ cursor: 'grab' }} {...listeners} {...attributes}>
-                        <DragIndicatorIcon/>
-                    </IconButton>
-                    </Box>
-                </CardActions>
-            </Card>
-            <Dialog
-                open={helpOpen}
-                onClose={() => setHelpOpen(false)}
-                maxWidth="lg"
-            >
-                <DialogTitle>
-                    {module.display_name}
-                </DialogTitle>
-                <DialogContent>
-                    <ReactMarkdown>
-                        {module.manifest.description.split("\n").map((line: string) => line.trim()).join("\n")}
-                    </ReactMarkdown>
-                </DialogContent>
-            </Dialog>
-            {module.configs && name != 'cli_feeder' && <ConfigPanel module={module} open={configOpen} setOpen={setConfigOpen} configValues={configValues} />}
-        </Grid>
-    )
-}
-
-function ConfigField({ config_value, module, configValues }: { config_value: any, module: Module, configValues: any }) {
-    const [showPassword, setShowPassword] = useState(false);
-    const handleClickShowPassword = () => setShowPassword((show) => !show);
-
-    const handleMouseDownPassword = (event: React.MouseEvent<HTMLButtonElement>) => {
-      event.preventDefault();
-    };
-  
-    const handleMouseUpPassword = (event: React.MouseEvent<HTMLButtonElement>) => {
-      event.preventDefault();
-    };
-
-    function setConfigValue(config: any, value: any) {
-        configValues[module.name][config] = value;
-    }
-    const config_args: Config = module.configs[config_value];
-    const config_name: string = config_value.replace(/_/g, " ");
-    const config_display_name = config_name.capitalize();
-    const value = configValues[module.name][config_value] || config_args.default;
-    
-
-    const config_value_lower = config_value.toLowerCase();
-    const is_password = config_value_lower.includes('password') ||
-                        config_value_lower.includes('secret') ||
-                        config_value_lower.includes('token') ||
-                        config_value_lower.includes('key') ||
-                        config_value_lower.includes('api_hash') ||
-                        config_args.type === 'password';
-
-    const text_input_type = is_password ? 'password' : (config_args.type === 'int' ? 'number' : 'text');
-
-    return (
-        <Box>
-            <Typography variant='body1' style={{ fontWeight: 'bold' }}>{config_display_name} {config_args.required && (`(required)`)} </Typography>
-            <FormControl size="small">
-                {config_args.type === 'bool' ?
-                    <FormControlLabel control={
-                        <Checkbox defaultChecked={value} size="small" id={`${module}.${config_value}`}
-                            onChange={(e) => {
-                                setConfigValue(config_value, e.target.checked);
-                            }}
-                        />} label={config_args.help.capitalize()}
-                    />
-                    :
-                    (
-                        config_args.choices !== undefined ?
-                            <Select size="small" id={`${module}.${config_value}`}
-                                defaultValue={config_args.default}
-                                value={value}
-                                onChange={(e) => {
-                                    setConfigValue(config_value, e.target.value);
-                                }}
-                            >
-                                {config_args.choices.map((choice: any) => {
-                                    return (
-                                        <MenuItem key={`${module}.${config_value}.${choice}`}
-                                            value={choice}>{choice}</MenuItem>
-                                    );
-                                })}
-                            </Select>
-                            :
-                            (config_args.type === 'json_loader' ?
-                                <TextField multiline size="small" id={`${module}.${config_value}`} defaultValue={JSON.stringify(value, null, 2)} rows={6} onChange={
-                                    (e) => {
-                                        try {
-                                            let val = JSON.parse(e.target.value);
-                                            setConfigValue(config_value, val);
-                                        } catch (e) {
-                                            console.log(e);
-                                        }
-                                    }
-                                } />
-                                :
-                                <TextField size="small" id={`${module}.${config_value}`} defaultValue={value} type={showPassword ? 'text' : text_input_type}
-                                    onChange={(e) => {
-                                        setConfigValue(config_value, e.target.value);
-                                    }}
-                                    required={config_args.required}
-                                    slotProps={ is_password ? {
-                                        input: { endAdornment: (
-                                            <InputAdornment position="end">
-                                                <IconButton
-                                                    aria-label="toggle password visibility"
-                                                    onClick={handleClickShowPassword}
-                                                    onMouseDown={handleMouseDownPassword}
-                                                    onMouseUp={handleMouseUpPassword}
-                                                >
-                                                    {showPassword ? <VisibilityOff /> : <Visibility />}
-                                                </IconButton>
-                                            </InputAdornment>
-                                        )}
-                                    } : {}}
-                                />
-                            )
-                    )
-                }
-                {config_args.type !== 'bool' && (
-                    <FormHelperText >{config_args.help.capitalize()}</FormHelperText>
-                )}
-            </FormControl>
-        </Box>
-    )
-}
-
-function ConfigPanel({ module, open, setOpen, configValues }: { module: Module, open: boolean, setOpen: any, configValues: any }) {
-
-    return (
-        <>
-            <Dialog
-                open={open}
-                onClose={() => setOpen(false)}
-                maxWidth="lg"
-            >
-                <DialogTitle>
-                    {module.display_name}
-                </DialogTitle>
-                <DialogContent>
-                    <Stack direction="column" spacing={1}>
-                        {Object.keys(module.configs).map((config_value: any) => {
-                            return (
-                                <ConfigField key={config_value} config_value={config_value} module={module} configValues={configValues} />
-                            );
-                        })}
-                    </Stack>
-                </DialogContent>
-            </Dialog>
-        </>
-    );
-}
-
-export default StepCard;
--- a/scripts/settings/src/main.tsx
+++ b/scripts/settings/src/main.tsx
@@ -1,44 +0,0 @@
-import * as React from 'react';
-import * as ReactDOM from 'react-dom/client';
-import { ThemeProvider } from '@mui/material/styles';
-import { CssBaseline } from '@mui/material';
-import App from './App';
-import { createTheme } from '@mui/material/styles';
-import { red } from '@mui/material/colors';
-import { useState, useEffect } from 'react';
-
-function RootApp() {
-  const [mode, setMode] = useState('light');
-
-useEffect(() => {
-    setMode(window.localStorage.getItem('theme') || 'light');
-}, []);
-
-var observer = new MutationObserver(function(mutations) {
-  setMode(window.localStorage.getItem('theme') || 'light');
-  
-})
-observer.observe(document.documentElement, {attributes: true, attributeFilter: ['data-theme']});
-
-// A custom theme for this app
-const theme = createTheme({
-  palette: {
-    mode: mode == 'light' ? 'light' : 'dark',
-  },
-  cssVariables: true
-});
-
-  return (
-    <ThemeProvider theme={theme}>
-    <CssBaseline />
-    <App  />
-  </ThemeProvider>
-  );
-}
-
-ReactDOM.createRoot(document.getElementById('root')!).render(
-  <React.StrictMode>
-    <RootApp />
-  </React.StrictMode>,
-);
-
--- a/scripts/settings/src/types.d.ts
+++ b/scripts/settings/src/types.d.ts
@@ -1,21 +0,0 @@
-export interface Config {
-  name: string;
-  description: string;
-  type: string?;
-  default: any;
-  help: string;
-  choices: string[];
-  required: boolean;
-}
-
-interface Manifest {
-  description: string;
-}
-
-export interface Module {
-  name: string;
-  description: string;
-  configs: { [key: string]: Config };
-  manifest: Manifest;
-  display_name: string;
-}
--- a/scripts/settings/tsconfig.json
+++ b/scripts/settings/tsconfig.json
@@ -1,21 +0,0 @@
-{
-  "compilerOptions": {
-    "target": "ESNext",
-    "useDefineForClassFields": true,
-    "lib": ["DOM", "DOM.Iterable", "ESNext"],
-    "allowJs": false,
-    "skipLibCheck": true,
-    "esModuleInterop": false,
-    "allowSyntheticDefaultImports": true,
-    "strict": true,
-    "forceConsistentCasingInFileNames": true,
-    "module": "ESNext",
-    "moduleResolution": "Node",
-    "resolveJsonModule": true,
-    "isolatedModules": true,
-    "noEmit": true,
-    "jsx": "react-jsx"
-  },
-  "include": ["src"],
-  "references": [{ "path": "./tsconfig.node.json" }]
-}
--- a/scripts/settings/tsconfig.node.json
+++ b/scripts/settings/tsconfig.node.json
@@ -1,9 +0,0 @@
-{
-  "compilerOptions": {
-    "composite": true,
-    "module": "ESNext",
-    "moduleResolution": "Node",
-    "allowSyntheticDefaultImports": true
-  },
-  "include": ["vite.config.ts"]
-}
--- a/scripts/settings/vite.config.ts
+++ b/scripts/settings/vite.config.ts
@@ -1,12 +0,0 @@
-import { defineConfig } from 'vite';
-import react from '@vitejs/plugin-react';
-import { viteSingleFile } from "vite-plugin-singlefile"
-
-// https://vite.dev/config/
-export default defineConfig({
-  plugins: [react(), viteSingleFile()],
-  build: {
-    // minify: false,
-    // sourcemap: true,
-  }
-});
--- a/scripts/telegram_setup.py
+++ b/scripts/telegram_setup.py
@@ -1,27 +0,0 @@
-"""
-This script is used to create a new session file for the Telegram client.
-To do this you must first create a Telegram application at https://my.telegram.org/apps
-And store your id and hash in the environment variables TELEGRAM_API_ID and TELEGRAM_API_HASH.
-Create a .env file, or add the following to your environment :
-```
-export TELEGRAM_API_ID=[YOUR_ID_HERE]
-export TELEGRAM_API_HASH=[YOUR_HASH_HERE]
-```
-Then run this script to create a new session file.
-
-You will need to provide your phone number and a 2FA code the first time you run this script.
-"""
-
-import os
-from telethon.sync import TelegramClient
-from loguru import logger
-
-
-# Create a
-API_ID = os.getenv("TELEGRAM_API_ID")
-API_HASH = os.getenv("TELEGRAM_API_HASH")
-SESSION_FILE = "secrets/anon-insta"
-
-os.makedirs("secrets", exist_ok=True)
-with TelegramClient(SESSION_FILE, API_ID, API_HASH) as client:
-    logger.success(f"New session file created: {SESSION_FILE}.session")
--- a/setup.cfg
+++ b/setup.cfg
@@ -0,0 +1,53 @@
+[metadata]
+name = auto_archiver
+version = attr: auto_archiver.version.__version__
+author = Bellingcat
+author_email = tech@bellingcat.com
+description = Easily archive online media content
+long_description = file: README.md
+long_description_content_type = text/markdown
+keywords = archive, oosi, osint, scraping
+license = MIT
+classifiers =
+	Intended Audience :: Developers
+	Intended Audience :: Science/Research
+	License :: OSI Approved :: MIT License
+	Programming Language :: Python :: 3
+project_urls = 
+	Source Code = https://github.com/bellingcat/auto-archiver
+	Bug Tracker = https://github.com/bellingcat/auto-archiver/issues
+	Bellingcat = https://www.bellingcat.com
+platforms = any
+
+[options]
+setup_requires =
+    setuptools-pipfile
+zip_safe = False
+package_dir=
+    =src
+packages=find:
+find_packages=true
+python_requires = >=3.8
+
+[options.package_data]
+* = *.html
+
+[options.entry_points]
+console_scripts =
+    auto-archiver = auto_archiver.__main__:main
+
+# [options.extras_require]
+# pdf = ReportLab>=1.2; RXP
+# rest = docutils>=0.3; pack ==1.1, ==1.3
+
+[options.packages.find]
+where=src
+# include=auto_archiver*
+# exclude =
+#     examples*
+#     .eggs*
+#     build*
+#     secrets*
+#     tmp*
+#     docs*
+#     src.tests*
--- a/setup.py
+++ b/setup.py
@@ -0,0 +1,4 @@
+from setuptools import setup
+
+if __name__ == "__main__":
+    setup()
--- a/src/auto_archiver/modules/html_formatter/templates/init.py
+++ b/src/auto_archiver/modules/html_formatter/templates/init.py
--- a/src/auto_archiver/init.py
+++ b/src/auto_archiver/init.py
@@ -0,0 +1,7 @@
+from . import archivers, databases, enrichers, feeders, formatters, storages, utils, core
+
+# need to manually specify due to cyclical deps
+from .core.orchestrator import ArchivingOrchestrator
+from .core.config import Config
+# making accessible directly
+from .core.metadata import Metadata
--- a/src/auto_archiver/main.py
+++ b/src/auto_archiver/main.py
@@ -1,12 +1,11 @@
-"""Entry point for the auto_archiver package."""
-
-from auto_archiver.core.orchestrator import ArchivingOrchestrator
-import sys
-
+from . import Config
+from . import ArchivingOrchestrator

 def main():
-    for _ in ArchivingOrchestrator()._command_line_run(sys.argv[1:]):
-        pass
+    config = Config()
+    config.parse()
+    orchestrator = ArchivingOrchestrator(config)
+    for r in orchestrator.feed(): pass


 if __name__ == "__main__":
--- a/src/auto_archiver/archivers/init.py
+++ b/src/auto_archiver/archivers/init.py
@@ -0,0 +1,10 @@
+from .archiver import Archiver
+from .telethon_archiver import TelethonArchiver
+from .twitter_archiver import TwitterArchiver
+from .twitter_api_archiver import TwitterApiArchiver
+from .instagram_archiver import InstagramArchiver
+from .instagram_tbot_archiver import InstagramTbotArchiver
+from .tiktok_archiver import TiktokArchiver
+from .telegram_archiver import TelegramArchiver
+from .vk_archiver import VkArchiver
+from .youtubedl_archiver import YoutubeDLArchiver
--- a/src/auto_archiver/archivers/archiver.py
+++ b/src/auto_archiver/archivers/archiver.py
@@ -0,0 +1,60 @@
+from __future__ import annotations
+from abc import abstractmethod
+from dataclasses import dataclass
+import os
+import mimetypes, requests
+
+from ..core import Metadata, Step, ArchivingContext
+
+
+@dataclass
+class Archiver(Step):
+    name = "archiver"
+
+    def __init__(self, config: dict) -> None:
+        # without this STEP.__init__ is not called
+        super().__init__(config)
+
+    def init(name: str, config: dict) -> Archiver:
+        # only for typing...
+        return Step.init(name, config, Archiver)
+
+    def setup(self) -> None:
+        # used when archivers need to login or do other one-time setup
+        pass
+
+    def sanitize_url(self, url: str) -> str:
+        # used to clean unnecessary URL parameters OR unfurl redirect links
+        return url
+
+    def _guess_file_type(self, path: str) -> str:
+        """
+        Receives a URL or filename and returns global mimetype like 'image' or 'video'
+        see https://developer.mozilla.org/en-US/docs/Web/HTTP/Basics_of_HTTP/MIME_types/Common_types
+        """
+        mime = mimetypes.guess_type(path)[0]
+        if mime is not None:
+            return mime.split("/")[0]
+        return ""
+
+    def download_from_url(self, url: str, to_filename: str = None, item: Metadata = None) -> str:
+        """
+        downloads a URL to provided filename, or inferred from URL, returns local filename, if item is present will use its tmp_dir
+        """
+        if not to_filename:
+            to_filename = url.split('/')[-1].split('?')[0]
+            if len(to_filename) > 64:
+                to_filename = to_filename[-64:]
+        if item:
+            to_filename = os.path.join(ArchivingContext.get_tmp_dir(), to_filename)
+        headers = {
+            'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/81.0.4044.138 Safari/537.36'
+        }
+        d = requests.get(url, headers=headers)
+        assert d.status_code == 200, f"got response code {d.status_code} for {url=}"
+        with open(to_filename, 'wb') as f:
+            f.write(d.content)
+        return to_filename
+
+    @abstractmethod
+    def download(self, item: Metadata) -> Metadata: pass
--- a/src/auto_archiver/archivers/instagram_archiver.py
+++ b/src/auto_archiver/archivers/instagram_archiver.py
@@ -0,0 +1,143 @@
+import re, os, shutil, traceback
+import instaloader  # https://instaloader.github.io/as-module.html
+from loguru import logger
+
+from . import Archiver
+from ..core import Metadata
+from ..core import Media
+
+class InstagramArchiver(Archiver):
+    """
+    Uses Instaloader to download either a post (inc images, videos, text) or as much as possible from a profile (posts, stories, highlights, ...)
+    """
+    name = "instagram_archiver"
+
+    # NB: post regex should be tested before profile
+    # https://regex101.com/r/MGPquX/1
+    post_pattern = re.compile(r"(?:(?:http|https):\/\/)?(?:www.)?(?:instagram.com|instagr.am|instagr.com)\/(?:p|reel)\/(\w+)")
+    # https://regex101.com/r/6Wbsxa/1
+    profile_pattern = re.compile(r"(?:(?:http|https):\/\/)?(?:www.)?(?:instagram.com|instagr.am|instagr.com)\/(\w+)")
+    # TODO: links to stories
+
+    def __init__(self, config: dict) -> None:
+        super().__init__(config)
+        # TODO: refactor how configuration validation is done
+        self.assert_valid_string("username")
+        self.assert_valid_string("password")
+        self.assert_valid_string("download_folder")
+        self.assert_valid_string("session_file")
+        self.insta = instaloader.Instaloader(
+            download_geotags=True, download_comments=True, compress_json=False, dirname_pattern=self.download_folder, filename_pattern="{date_utc}_UTC_{target}__{typename}"
+        )
+        try:
+            self.insta.load_session_from_file(self.username, self.session_file)
+        except Exception as e:
+            logger.error(f"Unable to login from session file: {e}\n{traceback.format_exc()}")
+            try:
+                self.insta.login(self.username, config.instagram_self.password)
+                # TODO: wait for this issue to be fixed https://github.com/instaloader/instaloader/issues/1758
+                self.insta.save_session_to_file(self.session_file)
+            except Exception as e2:
+                logger.error(f"Unable to finish login (retrying from file): {e2}\n{traceback.format_exc()}")
+
+    @staticmethod
+    def configs() -> dict:
+        return {
+            "username": {"default": None, "help": "a valid Instagram username"},
+            "password": {"default": None, "help": "the corresponding Instagram account password"},
+            "download_folder": {"default": "instaloader", "help": "name of a folder to temporarily download content to"},
+            "session_file": {"default": "secrets/instaloader.session", "help": "path to the instagram session which saves session credentials"},
+            #TODO: fine-grain
+            # "download_stories": {"default": True, "help": "if the link is to a user profile: whether to get stories information"},
+        }
+
+    def download(self, item: Metadata) -> Metadata:
+        url = item.get_url()
+
+        # detect URLs that we definitely cannot handle
+        post_matches = self.post_pattern.findall(url)
+        profile_matches = self.profile_pattern.findall(url)
+
+        # return if not a valid instagram link
+        if not len(post_matches) and not len(profile_matches): return
+
+        result = None
+        try:
+            os.makedirs(self.download_folder, exist_ok=True)
+            # process if post
+            if len(post_matches):
+                result = self.download_post(url, post_matches[0])
+            # process if profile
+            elif len(profile_matches):
+                result = self.download_profile(url, profile_matches[0])
+        except Exception as e:
+            logger.error(f"Failed to download with instagram archiver due to: {e}, make sure your account credentials are valid.")
+        finally:
+            shutil.rmtree(self.download_folder, ignore_errors=True)
+        return result
+
+    def download_post(self, url: str, post_id: str) -> Metadata:
+        logger.debug(f"Instagram {post_id=} detected in {url=}")
+
+        post = instaloader.Post.from_shortcode(self.insta.context, post_id)
+        if self.insta.download_post(post, target=post.owner_username):
+            return self.process_downloads(url, post.title, post._asdict(), post.date)
+
+    def download_profile(self, url: str, username: str) -> Metadata:
+        # gets posts, posts where username is tagged, igtv postss, stories, and highlights
+        logger.debug(f"Instagram {username=} detected in {url=}")
+
+        profile = instaloader.Profile.from_username(self.insta.context, username)
+        try:
+            for post in profile.get_posts():
+                try: self.insta.download_post(post, target=f"profile_post_{post.owner_username}")
+                except Exception as e: logger.error(f"Failed to download post: {post.shortcode}: {e}")
+        except Exception as e: logger.error(f"Failed profile.get_posts: {e}")
+
+        try:
+            for post in profile.get_tagged_posts():
+                try: self.insta.download_post(post, target=f"tagged_post_{post.owner_username}")
+                except Exception as e: logger.error(f"Failed to download tagged post: {post.shortcode}: {e}")
+        except Exception as e: logger.error(f"Failed profile.get_tagged_posts: {e}")
+
+        try:
+            for post in profile.get_igtv_posts():
+                try: self.insta.download_post(post, target=f"igtv_post_{post.owner_username}")
+                except Exception as e: logger.error(f"Failed to download igtv post: {post.shortcode}: {e}")
+        except Exception as e: logger.error(f"Failed profile.get_igtv_posts: {e}")
+
+        try:
+            for story in self.insta.get_stories([profile.userid]):
+                for item in story.get_items():
+                    try: self.insta.download_storyitem(item, target=f"story_item_{story.owner_username}")
+                    except Exception as e: logger.error(f"Failed to download story item: {item}: {e}")
+        except Exception as e: logger.error(f"Failed get_stories: {e}")
+
+        try:
+            for highlight in self.insta.get_highlights(profile.userid):
+                for item in highlight.get_items():
+                    try: self.insta.download_storyitem(item, target=f"highlight_item_{highlight.owner_username}")
+                    except Exception as e: logger.error(f"Failed to download highlight item: {item}: {e}")
+        except Exception as e: logger.error(f"Failed get_highlights: {e}")
+
+        return self.process_downloads(url, f"@{username}", profile._asdict(), None)
+
+    def process_downloads(self, url, title, content, date):
+        result = Metadata()
+        result.set_title(title).set_content(str(content)).set_timestamp(date)
+
+        try:
+            all_media = []
+            for f in os.listdir(self.download_folder):
+                if os.path.isfile((filename := os.path.join(self.download_folder, f))):
+                    if filename[-4:] == ".txt": continue
+                    all_media.append(Media(filename))
+
+            assert len(all_media) > 1, "No uploaded media found"
+            all_media.sort(key=lambda m: m.filename, reverse=True)
+            for m in all_media:
+                result.add_media(m)
+
+            return result.success("instagram")
+        except Exception as e:
+            logger.error(f"Could not fetch instagram post {url} due to: {e}")
--- a/src/auto_archiver/archivers/instagram_tbot_archiver.py
+++ b/src/auto_archiver/archivers/instagram_tbot_archiver.py
@@ -0,0 +1,77 @@
+
+from telethon.sync import TelegramClient
+from loguru import logger
+import time, os
+from sqlite3 import OperationalError
+from . import Archiver
+from ..core import Metadata, Media, ArchivingContext
+
+
+class InstagramTbotArchiver(Archiver):
+    """
+    calls a telegram bot to fetch instagram posts/stories... and gets available media from it
+    https://github.com/adw0rd/instagrapi
+    https://t.me/instagram_load_bot
+    """
+    name = "instagram_tbot_archiver"
+
+    def __init__(self, config: dict) -> None:
+        super().__init__(config)
+        self.assert_valid_string("api_id")
+        self.assert_valid_string("api_hash")
+        self.timeout = int(self.timeout)
+        try:
+            self.client = TelegramClient(self.session_file, self.api_id, self.api_hash)
+        except OperationalError as e:
+            logger.error(f"Unable to access the {self.session_file} session, please make sure you don't use the same session file here and in telethon_archiver. if you do then disable at least one of the archivers for the 1st time you setup telethon session: {e}")
+
+    @staticmethod
+    def configs() -> dict:
+        return {
+            "api_id": {"default": None, "help": "telegram API_ID value, go to https://my.telegram.org/apps"},
+            "api_hash": {"default": None, "help": "telegram API_HASH value, go to https://my.telegram.org/apps"},
+            "session_file": {"default": "secrets/anon-insta", "help": "optional, records the telegram login session for future usage, '.session' will be appended to the provided value."},
+            "timeout": {"default": 45, "help": "timeout to fetch the instagram content in seconds."},
+        }
+
+    def setup(self) -> None:
+        logger.info(f"SETUP {self.name} checking login...")
+        with self.client.start():
+            logger.success(f"SETUP {self.name} login works.")
+
+    def download(self, item: Metadata) -> Metadata:
+        url = item.get_url()
+        if not "instagram.com" in url: return False
+
+        result = Metadata()
+        tmp_dir = ArchivingContext.get_tmp_dir()
+        with self.client.start():
+            chat = self.client.get_entity("instagram_load_bot")
+            since_id = self.client.send_message(entity=chat, message=url).id
+
+            attempts = 0
+            seen_media = []
+            message = ""
+            time.sleep(3)
+            # media is added before text by the bot so it can be used as a stop-logic mechanism
+            while attempts < (self.timeout - 3) and (not message or not len(seen_media)):
+                attempts += 1
+                time.sleep(1)
+                for post in self.client.iter_messages(chat, min_id=since_id):
+                    since_id = max(since_id, post.id)
+                    if post.media and post.id not in seen_media:
+                        filename_dest = os.path.join(tmp_dir, f'{chat.id}_{post.id}')
+                        media = self.client.download_media(post.media, filename_dest)
+                        if media: 
+                            result.add_media(Media(media))
+                            seen_media.append(post.id)
+                    if post.message: message += post.message
+
+            if "You must enter a URL to a post" in message: 
+                logger.debug(f"invalid link {url=} for {self.name}: {message}")
+                return False
+                
+            if message:
+                result.set_content(message).set_title(message[:128])
+
+            return result.success("insta-via-bot")
--- a/src/auto_archiver/modules/telegram_extractor/telegram_extractor.py
+++ b/src/auto_archiver/modules/telegram_extractor/telegram_extractor.py
@@ -1,27 +1,32 @@
-import requests
-import re
-import html
+import requests, re, html
 from bs4 import BeautifulSoup
 from loguru import logger

-from auto_archiver.core import Extractor
-from auto_archiver.core import Metadata, Media
+from . import Archiver
+from ..core import Metadata, Media


-class TelegramExtractor(Extractor):
+class TelegramArchiver(Archiver):
    """
-    Extractor for telegram that does not require login, but the telethon_extractor is much more advised,
-    will only return if at least one image or one video is found
+    Archiver for telegram that does not require login, but the telethon_archiver is much more advised, will only return if at least one image or one video is found
    """
+    name = "telegram_archiver"
+
+    def __init__(self, config: dict) -> None:
+        super().__init__(config)
+
+    @staticmethod
+    def configs() -> dict:
+        return {}

    def download(self, item: Metadata) -> Metadata:
        url = item.get_url()
        # detect URLs that we definitely cannot handle
-        if "t.me" != item.netloc:
+        if 't.me' != item.netloc:
            return False

        headers = {
-            "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/81.0.4044.138 Safari/537.36"
+            'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/81.0.4044.138 Safari/537.36'
        }

        # TODO: check if we can do this more resilient to variable URLs
@@ -29,11 +34,11 @@ class TelegramExtractor(Extractor):
            url += "?embed=1"

        t = requests.get(url, headers=headers)
-        s = BeautifulSoup(t.content, "html.parser")
+        s = BeautifulSoup(t.content, 'html.parser')

        result = Metadata()
        result.set_content(html.escape(str(t.content)))
-        if timestamp := (s.find_all("time") or [{}])[0].get("datetime"):
+        if (timestamp := (s.find_all('time') or [{}])[0].get('datetime')):
            result.set_timestamp(timestamp)

        video = s.find("video")
@@ -43,26 +48,25 @@ class TelegramExtractor(Extractor):

            image_urls = []
            for im in image_tags:
-                urls = [u.replace("'", "") for u in re.findall(r"url\((.*?)\)", im["style"])]
+                urls = [u.replace("'", "") for u in re.findall(r'url\((.*?)\)', im['style'])]
                image_urls += urls

-            if not len(image_urls):
-                return False
+            if not len(image_urls): return False
            for img_url in image_urls:
-                result.add_media(Media(self.download_from_url(img_url)))
+                result.add_media(Media(self.download_from_url(img_url, item=item)))
        else:
-            video_url = video.get("src")
-            m_video = Media(self.download_from_url(video_url))
+            video_url = video.get('src')
+            m_video = Media(self.download_from_url(video_url, item=item))
            # extract duration from HTML
            try:
-                duration = s.find_all("time")[0].contents[0]
-                if ":" in duration:
-                    duration = float(duration.split(":")[0]) * 60 + float(duration.split(":")[1])
+                duration = s.find_all('time')[0].contents[0]
+                if ':' in duration:
+                    duration = float(duration.split(
+                        ':')[0]) * 60 + float(duration.split(':')[1])
                else:
                    duration = float(duration)
                m_video.set("duration", duration)
-            except Exception:
-                pass
+            except: pass
            result.add_media(m_video)

        return result.success("telegram")
--- a/src/auto_archiver/modules/telethon_extractor/telethon_extractor.py
+++ b/src/auto_archiver/modules/telethon_extractor/telethon_extractor.py
@@ -1,44 +1,49 @@
-import shutil
+
 from telethon.sync import TelegramClient
 from telethon.errors import ChannelInvalidError
 from telethon.tl.functions.messages import ImportChatInviteRequest
-from telethon.errors.rpcerrorlist import (
-    UserAlreadyParticipantError,
-    FloodWaitError,
-    InviteRequestSentError,
-    InviteHashExpiredError,
-)
+from telethon.errors.rpcerrorlist import UserAlreadyParticipantError, FloodWaitError, InviteRequestSentError, InviteHashExpiredError
 from loguru import logger
 from tqdm import tqdm
-import re
-import time
-import os
+import re, time, json, os

-from auto_archiver.core import Extractor
-from auto_archiver.core import Metadata, Media
-from auto_archiver.utils import random_str
+from . import Archiver
+from ..core import Metadata, Media, ArchivingContext


-class TelethonExtractor(Extractor):
-    valid_url = re.compile(r"https:\/\/t\.me(\/c){0,1}\/(.+)\/(\d+)")
+class TelethonArchiver(Archiver):
+    name = "telethon_archiver"
+    link_pattern = re.compile(r"https:\/\/t\.me(\/c){0,1}\/(.+)\/(\d+)")
    invite_pattern = re.compile(r"t.me(\/joinchat){0,1}\/\+?(.+)")

+    def __init__(self, config: dict) -> None:
+        super().__init__(config)
+        self.assert_valid_string("api_id")
+        self.assert_valid_string("api_hash")
+
+        self.client = TelegramClient(self.session_file, self.api_id, self.api_hash)
+
+    @staticmethod
+    def configs() -> dict:
+        return {
+            "api_id": {"default": None, "help": "telegram API_ID value, go to https://my.telegram.org/apps"},
+            "api_hash": {"default": None, "help": "telegram API_HASH value, go to https://my.telegram.org/apps"},
+            "bot_token": {"default": None, "help": "optional, but allows access to more content such as large videos, talk to @botfather"},
+            "session_file": {"default": "secrets/anon", "help": "optional, records the telegram login session for future usage, '.session' will be appended to the provided value."},
+            "join_channels": {"default": True, "help": "disables the initial setup with channel_invites config, useful if you have a lot and get stuck"},
+            "channel_invites": {
+                "default": {},
+                "help": "(JSON string) private channel invite links (format: t.me/joinchat/HASH OR t.me/+HASH) and (optional but important to avoid hanging for minutes on startup) channel id (format: CHANNEL_ID taken from a post url like https://t.me/c/CHANNEL_ID/1), the telegram account will join any new channels on setup",
+                "cli_set": lambda cli_val, cur_val: dict(cur_val, **json.loads(cli_val))
+            }
+        }
+
    def setup(self) -> None:
        """
-        1. makes a copy of session_file that is removed in cleanup
-        2. trigger login process for telegram or proceed if already saved in a session file
-        3. joins channel_invites where needed
+        1. trigger login process for telegram or proceed if already saved in a session file
+        2. joins channel_invites where needed
        """
        logger.info(f"SETUP {self.name} checking login...")
-
-        # make a copy of the session that is used exclusively with this archiver instance
-        new_session_file = os.path.join("secrets/", f"telethon-{time.strftime('%Y-%m-%d')}{random_str(8)}.session")
-        shutil.copy(self.session_file + ".session", new_session_file)
-        self.session_file = new_session_file.replace(".session", "")
-
-        # initiate the client
-        self.client = TelegramClient(self.session_file, self.api_id, self.api_hash)
-
        with self.client.start():
            logger.success(f"SETUP {self.name} login works.")

@@ -56,20 +61,18 @@ class TelethonExtractor(Extractor):
                    channel_invite = self.channel_invites[i]
                    channel_id = channel_invite.get("id", False)
                    invite = channel_invite["invite"]
-                    if match := self.invite_pattern.search(invite):
+                    if (match := self.invite_pattern.search(invite)):
                        try:
                            if channel_id:
                                ent = self.client.get_entity(int(channel_id))  # fails if not a member
                            else:
                                ent = self.client.get_entity(invite)  # fails if not a member
-                                logger.warning(
-                                    f"please add the property id='{ent.id}' to the 'channel_invites' configuration where {invite=}, not doing so can lead to a minutes-long setup time due to telegram's rate limiting."
-                                )
-                        except ValueError:
+                                logger.warning(f"please add the property id='{ent.id}' to the 'channel_invites' configuration where {invite=}, not doing so can lead to a minutes-long setup time due to telegram's rate limiting.")
+                        except ValueError as e:
                            logger.info(f"joining new channel {invite=}")
                            try:
                                self.client(ImportChatInviteRequest(match.group(2)))
-                            except UserAlreadyParticipantError:
+                            except UserAlreadyParticipantError as e:
                                logger.info(f"already joined {invite=}")
                            except InviteRequestSentError:
                                logger.warning(f"already sent a join request with {invite} still no answer")
@@ -86,12 +89,6 @@ class TelethonExtractor(Extractor):
                    i += 1
                    pbar.update()

-    def cleanup(self) -> None:
-        logger.info(f"CLEANUP {self.name}.")
-        session_file_name = self.session_file + ".session"
-        if os.path.exists(session_file_name):
-            os.remove(session_file_name)
-
    def download(self, item: Metadata) -> Metadata:
        """
        if this url is archivable will download post info and look for other posts from the same group with media.
@@ -99,10 +96,9 @@ class TelethonExtractor(Extractor):
        """
        url = item.get_url()
        # detect URLs that we definitely cannot handle
-        match = self.valid_url.search(url)
+        match = self.link_pattern.search(url)
        logger.debug(f"TELETHON: {match=}")
-        if not match:
-            return False
+        if not match: return False

        is_private = match.group(1) == "/c"
        chat = int(match.group(2)) if is_private else match.group(2)
@@ -112,56 +108,46 @@ class TelethonExtractor(Extractor):

        # NB: not using bot_token since then private channels cannot be archived: self.client.start(bot_token=self.bot_token)
        with self.client.start():
-            # with self.client.start(bot_token=self.bot_token):
+        # with self.client.start(bot_token=self.bot_token):
            try:
                post = self.client.get_messages(chat, ids=post_id)
            except ValueError as e:
                logger.error(f"Could not fetch telegram {url} possibly it's private: {e}")
                return False
            except ChannelInvalidError as e:
-                logger.error(
-                    f"Could not fetch telegram {url}. This error may be fixed if you setup a bot_token in addition to api_id and api_hash (but then private channels will not be archived, we need to update this logic to handle both): {e}"
-                )
+                logger.error(f"Could not fetch telegram {url}. This error may be fixed if you setup a bot_token in addition to api_id and api_hash (but then private channels will not be archived, we need to update this logic to handle both): {e}")
                return False

            logger.debug(f"TELETHON GOT POST {post=}")
-            if post is None:
-                return False
+            if post is None: return False

            media_posts = self._get_media_posts_in_group(chat, post)
-            logger.debug(f"got {len(media_posts)=} for {url=}")
+            logger.debug(f'got {len(media_posts)=} for {url=}')

-            tmp_dir = self.tmp_dir
+            tmp_dir = ArchivingContext.get_tmp_dir()

            group_id = post.grouped_id if post.grouped_id is not None else post.id
            title = post.message
            for mp in media_posts:
-                if len(mp.message) > len(title):
-                    title = mp.message  # save the longest text found (usually only 1)
+                if len(mp.message) > len(title): title = mp.message  # save the longest text found (usually only 1)

                # media can also be in entities
                if mp.entities:
-                    other_media_urls = [
-                        e.url
-                        for e in mp.entities
-                        if hasattr(e, "url") and e.url and self._guess_file_type(e.url) in ["video", "image", "audio"]
-                    ]
+                    other_media_urls = [e.url for e in mp.entities if hasattr(e, "url") and e.url and self._guess_file_type(e.url) in ["video", "image", "audio"]]
                    if len(other_media_urls):
                        logger.debug(f"Got {len(other_media_urls)} other media urls from {mp.id=}: {other_media_urls}")
                    for i, om_url in enumerate(other_media_urls):
-                        filename = self.download_from_url(om_url, f"{chat}_{group_id}_{i}")
+                        filename = self.download_from_url(om_url, f'{chat}_{group_id}_{i}', item)
                        result.add_media(Media(filename=filename), id=f"{group_id}_{i}")

-                filename_dest = os.path.join(tmp_dir, f"{chat}_{group_id}", str(mp.id))
+                filename_dest = os.path.join(tmp_dir, f'{chat}_{group_id}', str(mp.id))
                filename = self.client.download_media(mp.media, filename_dest)
                if not filename:
                    logger.debug(f"Empty media found, skipping {str(mp)=}")
                    continue
                result.add_media(Media(filename))

-            result.set_title(title).set_timestamp(post.date).set("api_data", post.to_dict())
-            if post.message != title:
-                result.set_content(post.message)
+            result.set_content(str(post)).set_title(title).set_timestamp(post.date)
        return result.success("telethon")

    def _get_media_posts_in_group(self, chat, original_post, max_amp=10):
--- a/src/auto_archiver/archivers/tiktok_archiver.py
+++ b/src/auto_archiver/archivers/tiktok_archiver.py
@@ -0,0 +1,56 @@
+import json, os, traceback
+import tiktok_downloader
+from loguru import logger
+
+
+from . import Archiver
+from ..core import Metadata, Media, ArchivingContext
+from ..utils.misc import random_str
+
+
+class TiktokArchiver(Archiver):
+    name = "tiktok_archiver"
+
+    def __init__(self, config: dict) -> None:
+        super().__init__(config)
+
+    @staticmethod
+    def configs() -> dict:
+        return {}
+
+    def download(self, item: Metadata) -> Metadata:
+        url = item.get_url()
+        if 'tiktok.com' not in url:
+            return False
+
+        result = Metadata()
+        try:
+            info = tiktok_downloader.info_post(url)
+            result.set_title(info.desc)
+            result.set_timestamp(info.create_time)
+            result.set_content(json.dumps({
+                "cover": info.cover,
+                "author": info.author,
+                "music_title": info.author,
+                "caption": getattr(info, "caption", info.desc),
+            }, ensure_ascii=False, indent=4))
+        except:
+            error = traceback.format_exc()
+            logger.warning(f'Other Tiktok error {error}')
+
+        try:
+            filename = os.path.join(ArchivingContext.get_tmp_dir(), f'{random_str(8)}.mp4')
+            tiktok_media = tiktok_downloader.snaptik(url).get_media()
+
+            if len(tiktok_media) <= 0:
+                logger.debug(f"TikTok: could not get media from {url=}")
+                return False
+
+            logger.info(f'downloading video {filename=}')
+            tiktok_media[0].download(filename)
+
+            result.add_media(Media(filename))
+            return result.success("tiktok")
+        except:
+            error = traceback.format_exc()
+            logger.warning(f'Other Tiktok error {error}')
--- a/src/auto_archiver/archivers/twitter_api_archiver.py
+++ b/src/auto_archiver/archivers/twitter_api_archiver.py
@@ -0,0 +1,98 @@
+
+import json, mimetypes
+from datetime import datetime
+from loguru import logger
+from pytwitter import Api
+from slugify import slugify
+
+from . import Archiver
+from .twitter_archiver import TwitterArchiver
+from ..core import Metadata,Media
+
+
+class TwitterApiArchiver(TwitterArchiver, Archiver):
+    name = "twitter_api_archiver"
+
+    def __init__(self, config: dict) -> None:
+        super().__init__(config)
+
+        if self.bearer_token:
+            self.assert_valid_string("bearer_token")
+            self.api = Api(bearer_token=self.bearer_token)
+        elif self.consumer_key and self.consumer_secret and self.access_token and self.access_secret:
+            self.assert_valid_string("consumer_key")
+            self.assert_valid_string("consumer_secret")
+            self.assert_valid_string("access_token")
+            self.assert_valid_string("access_secret")
+            self.api = Api(
+                consumer_key=self.consumer_key, consumer_secret=self.consumer_secret, access_token=self.access_token, access_secret=self.access_secret)
+        assert hasattr(self, "api") and self.api is not None, "Missing Twitter API configurations, please provide either bearer_token OR (consumer_key, consumer_secret, access_token, access_secret) to use this archiver."
+
+    @staticmethod
+    def configs() -> dict:
+        return {
+            "bearer_token": {"default": None, "help": "twitter API bearer_token which is enough for archiving, if not provided you will need consumer_key, consumer_secret, access_token, access_secret"},
+            "consumer_key": {"default": None, "help": "twitter API consumer_key"},
+            "consumer_secret": {"default": None, "help": "twitter API consumer_secret"},
+            "access_token": {"default": None, "help": "twitter API access_token"},
+            "access_secret": {"default": None, "help": "twitter API access_secret"},
+        }
+
+    def download(self, item: Metadata) -> Metadata:
+        url = item.get_url()
+        # detect URLs that we definitely cannot handle
+        username, tweet_id = self.get_username_tweet_id(url)
+        if not username: return False
+
+        try:
+            tweet = self.api.get_tweet(tweet_id, expansions=["attachments.media_keys"], media_fields=["type", "duration_ms", "url", "variants"], tweet_fields=["attachments", "author_id", "created_at", "entities", "id", "text", "possibly_sensitive"])
+        except Exception as e:
+            logger.error(f"Could not get tweet: {e}")
+            return False
+
+        result = Metadata()
+        result.set_title(tweet.data.text)
+        result.set_timestamp(datetime.strptime(tweet.data.created_at, "%Y-%m-%dT%H:%M:%S.%fZ"))
+
+        urls = []
+        if tweet.includes:
+            for i, m in enumerate(tweet.includes.media):
+                media = Media(filename="")
+                if m.url and len(m.url):
+                    media.set("src", m.url)
+                    media.set("duration", (m.duration_ms or 1) // 1000)
+                    mimetype = "image/jpeg"
+                elif hasattr(m, "variants"):
+                    variant = self.choose_variant(m.variants)
+                    if not variant: continue
+                    media.set("src", variant.url)
+                    mimetype = variant.content_type
+                else:
+                    continue
+                logger.info(f"Found media {media}")
+                ext = mimetypes.guess_extension(mimetype)
+                media.filename = self.download_from_url(media.get("src"), f'{slugify(url)}_{i}{ext}', item)
+                result.add_media(media)
+
+        result.set_content(json.dumps({
+            "id": tweet.data.id,
+            "text": tweet.data.text,
+            "created_at": tweet.data.created_at,
+            "author_id": tweet.data.author_id,
+            "geo": tweet.data.geo,
+            "lang": tweet.data.lang,
+            "media": urls
+        }, ensure_ascii=False, indent=4))
+        return result.success("twitter")
+
+    def choose_variant(self, variants):
+        # choosing the highest quality possible
+        variant, bit_rate = None, -1
+        for var in variants:
+            if var.content_type == "video/mp4":
+                if var.bit_rate > bit_rate:
+                    bit_rate = var.bit_rate
+                    variant = var
+            else:
+                variant = var if not variant else variant
+        return variant
--- a/src/auto_archiver/archivers/twitter_archiver.py
+++ b/src/auto_archiver/archivers/twitter_archiver.py
@@ -0,0 +1,152 @@
+import re, requests, mimetypes, json
+from datetime import datetime
+from loguru import logger
+from snscrape.modules.twitter import TwitterTweetScraper, Video, Gif, Photo
+from slugify import slugify
+
+from . import Archiver
+from ..core import Metadata, Media
+from ..utils import UrlUtil
+
+
+class TwitterArchiver(Archiver):
+    """
+    This Twitter Archiver uses unofficial scraping methods.
+    """
+
+    name = "twitter_archiver"
+    link_pattern = re.compile(r"(?:twitter|x).com\/(?:\#!\/)?(\w+)\/status(?:es)?\/(\d+)")
+    link_clean_pattern = re.compile(r"(.+(?:twitter|x)\.com\/.+\/\d+)(\?)*.*")
+
+    def __init__(self, config: dict) -> None:
+        super().__init__(config)
+
+    @staticmethod
+    def configs() -> dict:
+        return {}
+
+    def sanitize_url(self, url: str) -> str:
+        # expand URL if t.co and clean tracker GET params
+        if 'https://t.co/' in url:
+            try:
+                r = requests.get(url)
+                logger.debug(f'Expanded url {url} to {r.url}')
+                url = r.url
+            except:
+                logger.error(f'Failed to expand url {url}')
+        # https://twitter.com/MeCookieMonster/status/1617921633456640001?s=20&t=3d0g4ZQis7dCbSDg-mE7-w
+        return self.link_clean_pattern.sub("\\1", url)
+
+    def download(self, item: Metadata) -> Metadata:
+        """
+        if this url is archivable will download post info and look for other posts from the same group with media.
+        can handle private/public channels
+        """
+        url = item.get_url()
+        # detect URLs that we definitely cannot handle
+        username, tweet_id = self.get_username_tweet_id(url)
+        if not username: return False
+
+        result = Metadata()
+
+        scr = TwitterTweetScraper(tweet_id)
+        try:
+            tweet = next(scr.get_items())
+        except Exception as ex:
+            logger.warning(f"can't get tweet: {type(ex).__name__} occurred. args: {ex.args}")
+            return self.download_alternative(item, url, tweet_id)
+
+        result.set_title(tweet.content).set_content(tweet.json()).set_timestamp(tweet.date)
+        if tweet.media is None:
+            logger.debug(f'No media found, archiving tweet text only')
+            return result
+
+        for i, tweet_media in enumerate(tweet.media):
+            media = Media(filename="")
+            mimetype = ""
+            if type(tweet_media) == Video:
+                variant = max(
+                    [v for v in tweet_media.variants if v.bitrate], key=lambda v: v.bitrate)
+                media.set("src", variant.url).set("duration", tweet_media.duration)
+                mimetype = variant.contentType
+            elif type(tweet_media) == Gif:
+                variant = tweet_media.variants[0]
+                media.set("src", variant.url)
+                mimetype = variant.contentType
+            elif type(tweet_media) == Photo:
+                media.set("src", UrlUtil.twitter_best_quality_url(tweet_media.fullUrl))
+                mimetype = "image/jpeg"
+            else:
+                logger.warning(f"Could not get media URL of {tweet_media}")
+                continue
+            ext = mimetypes.guess_extension(mimetype)
+            media.filename = self.download_from_url(media.get("src"), f'{slugify(url)}_{i}{ext}', item)
+            result.add_media(media)
+
+        return result.success("twitter-snscrape")
+
+    def download_alternative(self, item: Metadata, url: str, tweet_id: str) -> Metadata:
+        """
+        Hack alternative working again.
+        https://stackoverflow.com/a/71867055/6196010 (OUTDATED URL)
+        https://github.com/JustAnotherArchivist/snscrape/issues/996#issuecomment-1615937362
+        next to test: https://cdn.embedly.com/widgets/media.html?&schema=twitter&url=https://twitter.com/bellingcat/status/1674700676612386816
+        """
+
+        logger.debug(f"Trying twitter hack for {url=}")
+        result = Metadata()
+
+        hack_url = f"https://cdn.syndication.twimg.com/tweet-result?id={tweet_id}"
+        r = requests.get(hack_url)
+        if r.status_code != 200: return False
+        tweet = r.json()
+
+        urls = []
+        for p in tweet.get("photos", []):
+            urls.append(p["url"])
+
+        # 1 tweet has 1 video max
+        if "video" in tweet:
+            v = tweet["video"]
+            urls.append(self.choose_variant(v.get("variants", [])))
+
+        logger.debug(f"Twitter hack got {urls=}")
+
+        for i, u in enumerate(urls):
+            media = Media(filename="")
+            u = UrlUtil.twitter_best_quality_url(u)
+            media.set("src", u)
+            ext = ""
+            if (mtype := mimetypes.guess_type(UrlUtil.remove_get_parameters(u))[0]):
+                ext = mimetypes.guess_extension(mtype)
+
+            media.filename = self.download_from_url(u, f'{slugify(url)}_{i}{ext}', item)
+            result.add_media(media)
+
+        result.set_title(tweet.get("text")).set_content(json.dumps(tweet, ensure_ascii=False)).set_timestamp(datetime.strptime(tweet["created_at"], "%Y-%m-%dT%H:%M:%S.%fZ"))
+        return result.success("twitter-hack")
+
+    def get_username_tweet_id(self, url):
+        # detect URLs that we definitely cannot handle
+        matches = self.link_pattern.findall(url)
+        if not len(matches): return False, False
+
+        username, tweet_id = matches[0]  # only one URL supported
+        logger.debug(f"Found {username=} and {tweet_id=} in {url=}")
+
+        return username, tweet_id
+
+    def choose_variant(self, variants):
+        # choosing the highest quality possible
+        variant, width, height = None, 0, 0
+        for var in variants:
+            if var.get("type", "") == "video/mp4":
+                width_height = re.search(r"\/(\d+)x(\d+)\/", var["src"])
+                if width_height:
+                    w, h = int(width_height[1]), int(width_height[2])
+                    if w > width or h > height:
+                        width, height = w, h
+                        variant = var.get("src", variant)
+            else:
+                variant = var.get("src") if not variant else variant
+        return variant
--- a/src/auto_archiver/modules/vk_extractor/vk_extractor.py
+++ b/src/auto_archiver/modules/vk_extractor/vk_extractor.py
@@ -1,30 +1,40 @@
 from loguru import logger
 from vk_url_scraper import VkScraper

-from auto_archiver.utils.misc import dump_payload
-from auto_archiver.core import Extractor
-from auto_archiver.core import Metadata, Media
+from ..utils.misc import dump_payload
+from . import Archiver
+from ..core import Metadata, Media, ArchivingContext


-class VkExtractor(Extractor):
-    """ "
+class VkArchiver(Archiver):
+    """"
    VK videos are handled by YTDownloader, this archiver gets posts text and images.
    Currently only works for /wall posts
    """
+    name = "vk_archiver"

-    def setup(self) -> None:
+    def __init__(self, config: dict) -> None:
+        super().__init__(config)
+        self.assert_valid_string("username")
+        self.assert_valid_string("password")
        self.vks = VkScraper(self.username, self.password, session_file=self.session_file)

+    @staticmethod
+    def configs() -> dict:
+        return {
+            "username": {"default": None, "help": "valid VKontakte username"},
+            "password": {"default": None, "help": "valid VKontakte password"},
+            "session_file": {"default": "secrets/vk_config.v2.json", "help": "valid VKontakte password"},
+        }
+
    def download(self, item: Metadata) -> Metadata:
        url = item.get_url()

-        if "vk.com" not in item.netloc:
-            return False
+        if "vk.com" not in item.netloc: return False

        # some urls can contain multiple wall/photo/... parts and all will be fetched
        vk_scrapes = self.vks.scrape(url)
-        if not len(vk_scrapes):
-            return False
+        if not len(vk_scrapes): return False
        logger.debug(f"VK: got {len(vk_scrapes)} scraped instances")

        result = Metadata()
@@ -36,7 +46,7 @@ class VkExtractor(Extractor):

        result.set_content(dump_payload(vk_scrapes))

-        filenames = self.vks.download_media(vk_scrapes, self.tmp_dir)
+        filenames = self.vks.download_media(vk_scrapes, ArchivingContext.get_tmp_dir())
        for filename in filenames:
            result.add_media(Media(filename))

--- a/Show More
+++ b/Show More