Remove unittest and switch to pytest fully

Add documentation on running tests to the readme
Revert change to twitter_archiver
2026-06-10 20:28:28 +03:00 · 2025-01-14 16:28:39 +01:00 · 2025-01-14 11:30:06 +01:00 · 2025-01-14 10:30:41 +01:00 · 2025-01-14 10:24:14 +01:00 · 2025-01-13 20:42:08 +01:00
40 changed files with 4221 additions and 2416 deletions
--- a/.github/workflows/python-publish.yaml
+++ b/.github/workflows/python-publish.yaml
@@ -1,4 +1,4 @@
-# This workflow will upload a Python Package using Twine when a release is created
+# This workflow uploads a Python Package to PyPI using Poetry when a release is created
 # For more information see: https://help.github.com/en/actions/language-and-framework-guides/using-python-with-github-actions#publishing-to-package-registries

 # This workflow uses actions that are not certified by GitHub.
@@ -21,30 +21,34 @@ jobs:
    runs-on: ubuntu-latest

    steps:
-    - uses: actions/checkout@v3
+    - name: Checkout Repository
+      uses: actions/checkout@v3

-    - name: Set up Python 3.10
+    - name: Extract Python Version from pyproject.toml
+      id: python-version
+      run: |
+        version=$(grep 'python =' pyproject.toml | awk -F'"' '{print $2}' | tr -d '^~<=>')
+        echo "python-version=$version" >> $GITHUB_ENV
+
+    - name: Set up Python
      uses: actions/setup-python@v4
      with:
-        python-version: "3.10"
+        python-version: ${{ env.python-version }}
+
+    - name: Install Poetry
+      run: |
+        python -m pip install --upgrade pip
+        python -m pip install "poetry>=2.0.0,<3.0.0"

    - name: Install dependencies
      run: |
-        python -m pip install --upgrade --upgrade-strategy=eager pip setuptools wheel twine pipenv
-        python -m pip install -e . --upgrade
-        python -m pipenv install --dev --python 3.10
-      env:
-        PIPENV_DEFAULT_PYTHON_VERSION: "3.10"
+        poetry install --no-root

-    - name: Build wheels
+    - name: Build the package
      run: |
-        python -m pipenv run python setup.py sdist bdist_wheel
+        poetry build

-    - name: Publish a Python distribution to PyPI
-      uses: pypa/gh-action-pypi-publish@release/v1
-      with:
-        user: __token__
-        verbose: true
-        skip_existing: true
-        password: ${{ secrets.PYPI_API_TOKEN }}
-        packages_dir: dist/
+    # Step 6: Publish to PyPI
+    - name: Publish to PyPI
+      run: |
+        poetry publish --username __token__ --password ${{ secrets.PYPI_API_TOKEN }}
--- a/.github/workflows/tests-core.yaml
+++ b/.github/workflows/tests-core.yaml
@@ -0,0 +1,38 @@
+name: Core Tests
+
+on:
+  push:
+    branches: [ main ]
+    paths:
+      - src/**
+  pull_request:
+    paths:
+      - src/**
+
+jobs:
+  tests:
+    runs-on: ubuntu-22.04
+    strategy:
+      matrix:
+        python-version: ["3.10", "3.11", "3.12"]
+    defaults:
+      run:
+        working-directory: ./
+
+    steps:
+      - uses: actions/checkout@v4
+
+      - name: Install Poetry
+        run: pipx install poetry
+
+      - name: Set up Python ${{ matrix.python-version }}
+        uses: actions/setup-python@v5
+        with:
+          python-version: ${{ matrix.python-version }}
+          cache: 'poetry'
+
+      - name: Install dependencies
+        run: poetry install --no-interaction --with dev
+
+      - name: Run Core Tests
+        run: poetry run pytest -ra -v -m "not download"
--- a/.github/workflows/tests-download.yaml
+++ b/.github/workflows/tests-download.yaml
@@ -0,0 +1,38 @@
+name: Download Tests
+
+on:
+  schedule:
+    - cron: '35 14 * * 1'
+  pull_request:
+    branches: [ main ]
+    paths:
+      - src/**
+
+jobs:
+  tests:
+    runs-on: ubuntu-22.04
+    strategy:
+      fail-fast: false
+      matrix:
+        python-version: ["3.10"] # only run expensive downloads on one (lowest) python version
+    defaults:
+      run:
+        working-directory: ./
+
+    steps:
+      - uses: actions/checkout@v4
+
+      - name: Install poetry
+        run: pipx install poetry
+
+      - name: Set up Python ${{ matrix.python-version }}
+        uses: actions/setup-python@v5
+        with:
+          python-version: ${{ matrix.python-version }}
+          cache: 'poetry'
+
+      - name: Install dependencies
+        run: poetry install --no-interaction --with dev
+
+      - name: Run Download Tests
+        run: poetry run pytest -ra -v -m "download"
--- a/.gitignore
+++ b/.gitignore
@@ -29,3 +29,4 @@ auto_archiver.egg-info*
 logs*
 *.csv
 archived/
+dist*
--- a/66
+++ b/66
@@ -1,30 +1,58 @@
-FROM webrecorder/browsertrix-crawler:1.0.4
+FROM webrecorder/browsertrix-crawler:1.0.4 AS base

-ENV RUNNING_IN_DOCKER=1
+ENV RUNNING_IN_DOCKER=1 \
+    LANG=C.UTF-8 \
+    LC_ALL=C.UTF-8 \
+    PYTHONDONTWRITEBYTECODE=1 \
+    PYTHONFAULTHANDLER=1 \
+    PATH="/root/.local/bin:$PATH"
+
+# Installing system dependencies
+RUN add-apt-repository ppa:mozillateam/ppa && \
+	apt-get update && \
+    apt-get install -y --no-install-recommends gcc ffmpeg fonts-noto exiftool && \
+	apt-get install -y --no-install-recommends firefox-esr && \
+    ln -s /usr/bin/firefox-esr /usr/bin/firefox && \
+    wget https://github.com/mozilla/geckodriver/releases/download/v0.33.0/geckodriver-v0.33.0-linux64.tar.gz && \
+    tar -xvzf geckodriver* -C /usr/local/bin && \
+    chmod +x /usr/local/bin/geckodriver && \
+    rm geckodriver-v* && \
+    apt-get clean && \
+    rm -rf /var/lib/apt/lists/*
+
+
+# Poetry and runtime
+FROM base AS runtime
+
+ENV POETRY_NO_INTERACTION=1 \
+    POETRY_VIRTUALENVS_IN_PROJECT=1 \
+    POETRY_VIRTUALENVS_CREATE=1
+
+
+RUN pip install --upgrade pip && \
+    pip install "poetry>=2.0.0,<3.0.0"

 WORKDIR /app

-RUN pip install --upgrade pip && \
-	pip install pipenv && \
-	add-apt-repository ppa:mozillateam/ppa && \
-	apt-get update && \
-	apt-get install -y gcc ffmpeg fonts-noto exiftool && \
-	apt-get install -y --no-install-recommends firefox-esr && \
-	ln -s /usr/bin/firefox-esr /usr/bin/firefox && \
-	wget https://github.com/mozilla/geckodriver/releases/download/v0.33.0/geckodriver-v0.33.0-linux64.tar.gz && \
-	tar -xvzf geckodriver* -C /usr/local/bin && \
-	chmod +x /usr/local/bin/geckodriver && \
-	rm geckodriver-v*
+
+COPY pyproject.toml poetry.lock README.md ./
+# Copy dependency files and install dependencies (excluding the package itself)
+RUN poetry install --only main --no-root --no-cache


-COPY Pipfile* ./
-# install from pipenv, with browsertrix-only requirements
-RUN pipenv install
+# Copy code: This is needed for poetry to install the package itself,
+# but the environment should be cached from the previous step if toml and lock files haven't changed
+COPY ./src/ .
+RUN poetry install --only main --no-cache

-# doing this at the end helps during development, builds are quick
-COPY ./src/ . 

-ENTRYPOINT ["pipenv", "run", "python3", "-m", "auto_archiver"]
+# Update PATH to include virtual environment binaries
+# Allowing entry point to run the application directly with Python
+ENV VIRTUAL_ENV=/app/.venv \
+    PATH="/app/.venv/bin:$PATH"
+
+ENTRYPOINT ["python3", "-m", "auto_archiver"]

 # should be executed with 2 volumes (3 if local_storage is used)
 # docker run --rm -v $PWD/secrets:/app/secrets -v $PWD/local_archive:/app/local_archive aa pipenv run python3 -m auto_archiver --config secrets/orchestration.yaml
+
--- a/50
+++ b/50
@@ -1,50 +0,0 @@
-[[source]]
-url = "https://pypi.org/simple"
-verify_ssl = true
-name = "pypi"
-
-[packages]
-gspread = "*"
-boto3 = "*"
-argparse = "*"
-beautifulsoup4 = "*"
-tiktok-downloader = "*"
-bs4 = "*"
-loguru = "*"
-ffmpeg-python = "*"
-selenium = "*"
-snscrape = "*"
-telethon = "*"
-google-api-python-client = "*"
-google-auth-httplib2 = "*"
-google-auth-oauthlib = "*"
-oauth2client = "*"
-pdqhash = "*"
-pillow = "*"
-python-slugify = "*"
-pyyaml = "*"
-dateparser = "*"
-python-twitter-v2 = "*"
-instaloader = "*"
-tqdm = "*"
-jinja2 = "*"
-cryptography = "*"
-dataclasses-json = "*"
-yt-dlp = "*"
-vk-url-scraper = "*"
-requests = {extras = ["socks"], version = "*"}
-numpy = "*"
-warcio = "*"
-jsonlines = "*"
-pysubs2 = "*"
-minify-html = "*"
-retrying = "*"
-tsp-client = "*"
-certvalidator = "*"
-
-[dev-packages]
-autopep8 = "*"
-setuptools-pipfile = "*"
-
-[requires]
-python_version = "3.10"
--- a/Pipfile.lock
+++ b/Pipfile.lock
--- a/README.md
+++ b/README.md
@@ -2,6 +2,8 @@

 [![PyPI version](https://badge.fury.io/py/auto-archiver.svg)](https://badge.fury.io/py/auto-archiver)
 [![Docker Image Version (latest by date)](https://img.shields.io/docker/v/bellingcat/auto-archiver?label=version&logo=docker)](https://hub.docker.com/r/bellingcat/auto-archiver)
+[![Core Test Status](https://github.com/bellingcat/auto-archiver/workflows/Core%20Tests/badge.svg)](https://github.com/bellingcat/auto-archiver/actions/workflows/tests-core.yaml)
+[![Download Test Status](https://github.com/bellingcat/auto-archiver/workflows/Download%20Tests/badge.svg)](https://github.com/bellingcat/auto-archiver/actions/workflows/tests-download.yaml)
 <!-- ![Docker Pulls](https://img.shields.io/docker/pulls/bellingcat/auto-archiver) -->
 <!-- [![PyPI download month](https://img.shields.io/pypi/dm/auto-archiver.svg)](https://pypi.python.org/pypi/auto-archiver/) -->
 <!-- [![Documentation Status](https://readthedocs.org/projects/vk-url-scraper/badge/?version=latest)](https://vk-url-scraper.readthedocs.io/en/latest/?badge=latest) -->
@@ -47,8 +49,8 @@ Docker works like a virtual machine running inside your computer, it isolates ev

 <details><summary><code>Python package instructions</code></summary>

-1. make sure you have python 3.8 or higher installed
-2. install the package `pip/pipenv/conda install auto-archiver`
+1. make sure you have python 3.10 or higher installed
+2. install the package with your preferred package manager: `pip/pipenv/conda install auto-archiver` or `poetry add auto-archiver`
 3. test it's installed with `auto-archiver --help`
 4. run it with your orchestration file and pass any flags you want in the command line `auto-archiver --config secrets/orchestration.yaml` if your orchestration file is inside a `secrets/`, which we advise
   
@@ -66,12 +68,13 @@ This can also be used for development.
 Install the following locally:
 1. [ffmpeg](https://www.ffmpeg.org/) must also be installed locally for this tool to work. 
 2. [firefox](https://www.mozilla.org/en-US/firefox/new/) and [geckodriver](https://github.com/mozilla/geckodriver/releases) on a path folder like `/usr/local/bin`. 
-3. (optional) [fonts-noto](https://fonts.google.com/noto) to deal with multiple unicode characters during selenium/geckodriver's screenshots: `sudo apt install fonts-noto -y`. 
+3. [Poetry](https://python-poetry.org/docs/#installation) for dependency management and packaging.
+4. (optional) [fonts-noto](https://fonts.google.com/noto) to deal with multiple unicode characters during selenium/geckodriver's screenshots: `sudo apt install fonts-noto -y`.

 Clone and run:
 1. `git clone https://github.com/bellingcat/auto-archiver`
-2. `pipenv install`
-3. `pipenv run python -m src.auto_archiver --config secrets/orchestration.yaml`
+2. `poetry install`
+3. `poetry run python -m src.auto_archiver --config secrets/orchestration.yaml`


 </details><br/>
@@ -108,7 +111,7 @@ configurations:
  # ... configurations for the other steps here ...
 ```

-To see all available `steps` (which archivers, storages, databses, ...) exist check the [example.orchestration.yaml](example.orchestration.yaml).
+To see all available `steps` (which archivers, storages, databases, ...) exist check the [example.orchestration.yaml](example.orchestration.yaml).

 All the `configurations` in the `orchestration.yaml` file (you can name it differently but need to pass it in the `--config FILENAME` argument) can be seen in the console by using the `--help` flag. They can also be overwritten, for example if you are using the `cli_feeder` to archive from the command line and want to provide the URLs you should do:

@@ -117,7 +120,7 @@ auto-archiver --config secrets/orchestration.yaml --cli_feeder.urls="url1,url2,u
 ```

 Here's the complete workflow that the auto-archiver goes through:
-```mermaid
+```{mermaid}
 graph TD
    s((start)) --> F(fa:fa-table Feeder)
    F -->|get and clean URL| D1{fa:fa-database Database}
@@ -258,6 +261,20 @@ The "archive location" link contains the path of the archived file, in local sto
 ## Development
 Use `python -m src.auto_archiver --config secrets/orchestration.yaml` to run from the local development environment.

+### Testing
+
+Tests are split using `pytest.mark` into 'core' and 'download' tests. Download tests will hit the network and make API calls (e.g. Twitter, Bluesky etc.) and should be run regularly to make sure that APIs have not changed.
+
+Tests can be run as follows:
+```
+# run core tests
+pytest -ra -v -m "not download" # or poetry run pytest -ra -v -m "not download"
+# run download tests
+pytest -ra -v -m "download" # or poetry run pytest -ra -v -m "download"
+# run all tests
+pytest -ra -v # or poetry run pytest -ra -v
+```
+
 #### Docker development
 working with docker locally:
  * `docker build . -t auto-archiver` to build a local image
--- a/example.orchestration.yaml
+++ b/example.orchestration.yaml
@@ -2,6 +2,7 @@ steps:
  # only 1 feeder allowed
  feeder: gsheet_feeder # defaults to cli_feeder
  archivers: # order matters, uncomment to activate
+    - bluesky_archiver
    # - vk_archiver
    # - telethon_archiver
    # - telegram_archiver
@@ -16,8 +17,13 @@ steps:
    # - wacz_archiver_enricher
  enrichers:
    - hash_enricher
+    # - meta_enricher
    # - metadata_enricher
    # - screenshot_enricher
+    # - pdq_hash_enricher
+    # - ssl_enricher
+    # - timestamping_enricher
+    # - whisper_enricher
    # - thumbnail_enricher
    # - wayback_archiver_enricher
    # - wacz_archiver_enricher
@@ -89,9 +95,33 @@ configurations:
    password: "vk pass"
    session_file: "secrets/vk_config.v2.json"

+  youtubedl_archiver:
+    subtitles: true
+    # use one of the following two methods to authenticate in youtube - either provide a cookies file or use the cookies of the given browser
+    # for more information, see https://github.com/yt-dlp/yt-dlp/wiki/FAQ#how-do-i-pass-cookies-to-yt-dlp
+    # cookie_file: "secrets/youtube_cookies.txt"
+    # cookies_from_browser: firefox
+    # proxy: socks5://proxy-user:password@proxy-ip:port
+
  screenshot_enricher:
    width: 1280
    height: 2300
+    # to save as pdf, uncomment the following lines and adjust the print options
+    # save_to_pdf: true
+    # print_options:
+      # for all options see https://www.selenium.dev/selenium/docs/api/py/webdriver/selenium.webdriver.common.print_page_options.html
+      # background: true
+      # orientation: "portrait"
+      # scale: 1
+      # page_width: 8.5in
+      # page_height: 11in
+      # margin_top: 0.4in
+      # margin_bottom: 0.4in
+      # margin_left: 0.4in
+      # margin_right: 0.4in
+      # page_ranges: ""
+      # shrink_to_fit: true
+
  wayback_archiver_enricher:
    timeout: 10
    key: "wayback key"
--- a/poetry.lock
+++ b/poetry.lock
--- a/pyproject.toml
+++ b/pyproject.toml
@@ -1,4 +1,81 @@
 [build-system]
-requires = ["setuptools", "wheel", "setuptools-pipfile"]
-build-backend = "setuptools.build_meta"
-[tool.setuptools-pipfile]
+requires = ["poetry-core>=2.0.0,<3.0.0"]
+build-backend = "poetry.core.masonry.api"
+
+[project]
+name = "auto-archiver"
+version = "0.13.0"
+description = "Automatically archive links to videos, images, and social media content from Google Sheets (and more)."
+
+requires-python = ">=3.10,<3.13"
+license = "MIT"
+authors = [
+    { name = "Bellingcat", email = "tech@bellingcat.com" },
+]
+readme = "README.md"
+keywords = ["archive", "oosi", "osint", "scraping"]
+classifiers = [
+    "Intended Audience :: Developers",
+    "Intended Audience :: Science/Research",
+    "License :: OSI Approved :: MIT License",
+    "Programming Language :: Python :: 3"
+]
+
+dependencies = [
+    "gspread (>=0.0.0)",
+    "argparse (>=0.0.0)",
+    "beautifulsoup4 (>=0.0.0)",
+    "tiktok-downloader (>=0.0.0)",
+    "bs4 (>=0.0.0)",
+    "loguru (>=0.0.0)",
+    "ffmpeg-python (>=0.0.0)",
+    "selenium (>=0.0.0)",
+    "telethon (>=0.0.0)",
+    "google-api-python-client (>=0.0.0)",
+    "google-auth-httplib2 (>=0.0.0)",
+    "google-auth-oauthlib (>=0.0.0)",
+    "oauth2client (>=0.0.0)",
+    "pdqhash (>=0.0.0)",
+    "pillow (>=0.0.0)",
+    "python-slugify (>=0.0.0)",
+    "pyyaml (>=0.0.0)",
+    "dateparser (>=0.0.0)",
+    "python-twitter-v2 (>=0.0.0)",
+    "instaloader (>=0.0.0)",
+    "tqdm (>=0.0.0)",
+    "jinja2 (>=0.0.0)",
+    "pyOpenSSL (==24.2.1)",
+    "cryptography (>=41.0.0,<42.0.0)",
+    "boto3 (>=1.28.0,<2.0.0)",
+    "dataclasses-json (>=0.0.0)",
+    "yt-dlp (==2024.09.27)",
+    "numpy (==2.1.3)",
+    "vk-url-scraper (>=0.0.0)",
+    "requests[socks] (>=0.0.0)",
+    "warcio (>=0.0.0)",
+    "jsonlines (>=0.0.0)",
+    "pysubs2 (>=0.0.0)",
+    "minify-html (>=0.0.0)",
+    "retrying (>=0.0.0)",
+    "tsp-client (>=0.0.0)",
+    "certvalidator (>=0.0.0)",
+    "toml (>=0.10.2,<0.11.0)"
+]
+
+[tool.poetry.group.dev.dependencies]
+pytest = "^8.3.4"
+autopep8 = "^2.3.1"
+
+[project.scripts]
+auto-archiver = "auto_archiver.__main__:main"
+
+[project.urls]
+homepage = "https://github.com/bellingcat/auto-archiver"
+repository = "https://github.com/bellingcat/auto-archiver"
+documentation = "https://github.com/bellingcat/auto-archiver"
+
+
+[tool.pytest.ini_options]
+markers = [
+    "download: marks tests that download content from the network",
+]
--- a/setup.cfg
+++ b/setup.cfg
@@ -1,53 +0,0 @@
-[metadata]
-name = auto_archiver
-version = attr: auto_archiver.version.__version__
-author = Bellingcat
-author_email = tech@bellingcat.com
-description = Easily archive online media content
-long_description = file: README.md
-long_description_content_type = text/markdown
-keywords = archive, oosi, osint, scraping
-license = MIT
-classifiers =
-	Intended Audience :: Developers
-	Intended Audience :: Science/Research
-	License :: OSI Approved :: MIT License
-	Programming Language :: Python :: 3
-project_urls = 
-	Source Code = https://github.com/bellingcat/auto-archiver
-	Bug Tracker = https://github.com/bellingcat/auto-archiver/issues
-	Bellingcat = https://www.bellingcat.com
-platforms = any
-
-[options]
-setup_requires =
-    setuptools-pipfile
-zip_safe = False
-package_dir=
-    =src
-packages=find:
-find_packages=true
-python_requires = >=3.8
-
-[options.package_data]
-* = *.html
-
-[options.entry_points]
-console_scripts =
-    auto-archiver = auto_archiver.__main__:main
-
-# [options.extras_require]
-# pdf = ReportLab>=1.2; RXP
-# rest = docutils>=0.3; pack ==1.1, ==1.3
-
-[options.packages.find]
-where=src
-# include=auto_archiver*
-# exclude =
-#     examples*
-#     .eggs*
-#     build*
-#     secrets*
-#     tmp*
-#     docs*
-#     src.tests*
--- a/setup.py
+++ b/setup.py
@@ -1,4 +0,0 @@
-from setuptools import setup
-
-if __name__ == "__main__":
-    setup()
--- a/src/auto_archiver/archivers/init.py
+++ b/src/auto_archiver/archivers/init.py
@@ -8,4 +8,5 @@ from .tiktok_archiver import TiktokArchiver
 from .telegram_archiver import TelegramArchiver
 from .vk_archiver import VkArchiver
 from .youtubedl_archiver import YoutubeDLArchiver
-from .instagram_api_archiver import InstagramAPIArchiver
+from .instagram_api_archiver import InstagramAPIArchiver
+from .bluesky_archiver import BlueskyArchiver
--- a/src/auto_archiver/archivers/archiver.py
+++ b/src/auto_archiver/archivers/archiver.py
@@ -48,6 +48,8 @@ class Archiver(Step):
        """
        downloads a URL to provided filename, or inferred from URL, returns local filename
        """
+        # TODO: should we refactor to use requests.get(url, stream=True) and write to file in chunks? compare approaches
+        # TODO: should we guess the extension?
        if not to_filename:
            to_filename = url.split('/')[-1].split('?')[0]
            if len(to_filename) > 64:
--- a/src/auto_archiver/archivers/bluesky_archiver.py
+++ b/src/auto_archiver/archivers/bluesky_archiver.py
@@ -0,0 +1,119 @@
+import os
+import re, requests, mimetypes
+from loguru import logger
+
+
+from . import Archiver
+from ..core import Metadata, Media, ArchivingContext
+
+
+class BlueskyArchiver(Archiver):
+    """
+    Uses an unauthenticated Bluesky API to archive posts including metadata, images and videos. Relies on `public.api.bsky.app/xrpc` and `bsky.social/xrpc`. Avoids ATProto to avoid auth.
+
+    Some inspiration from https://github.com/yt-dlp/yt-dlp/blob/master/yt_dlp/extractor/bluesky.py
+    """
+    name = "bluesky_archiver"
+    BSKY_POST = re.compile(r"/profile/([^/]+)/post/([a-zA-Z0-9]+)")
+
+    def __init__(self, config: dict) -> None:
+        super().__init__(config)
+
+    @staticmethod
+    def configs() -> dict:
+        return {}
+
+    def download(self, item: Metadata) -> Metadata:
+        url = item.get_url()
+        if not re.search(self.BSKY_POST, url):
+            return False
+
+        logger.debug(f"Identified a Bluesky post: {url}, archiving...")
+        result = Metadata()
+
+        # fetch post info and update result
+        post = self._get_post_from_uri(url)
+        logger.debug(f"Extracted post info: {post['record']['text']}")
+        result.set_title(post["record"]["text"])
+        result.set_timestamp(post["record"]["createdAt"])
+        for k, v in self._get_post_data(post).items():
+            if v: result.set(k, v)
+
+        # download if embeds present (1 video XOR >=1 images)
+        for media in self._download_bsky_embeds(post):
+            result.add_media(media)
+        logger.debug(f"Downloaded {len(result.media)} media files")
+
+        return result.success("bluesky")
+
+    def _get_post_from_uri(self, post_uri: str) -> dict:
+        """
+        Calls a public (no auth needed) Bluesky API to get a post from its uri, uses .getPostThread as it brings author info as well (unlike .getPost).
+        """
+        post_match = re.search(self.BSKY_POST, post_uri)
+        username = post_match.group(1)
+        post_id = post_match.group(2)
+        at_uri = f'at://{username}/app.bsky.feed.post/{post_id}'
+        r = requests.get(f"https://public.api.bsky.app/xrpc/app.bsky.feed.getPostThread?uri={at_uri}&depth=0&parent_height=0")
+        r.raise_for_status()
+        thread = r.json()
+        assert thread["thread"]["$type"] == "app.bsky.feed.defs#threadViewPost"
+        return thread["thread"]["post"]
+
+    def _download_bsky_embeds(self, post: dict) -> list[Media]:
+        """
+        Iterates over image(s) or video in a Bluesky post and downloads them        
+        """
+        media = []
+        embed = post.get("record", {}).get("embed", {})
+        image_medias = embed.get("images", []) + embed.get("media", {}).get("images", [])
+        video_medias = [e for e in [embed.get("video"), embed.get("media", {}).get("video")] if e]
+
+        for image_media in image_medias:
+                image_media = self._download_bsky_file_as_media(image_media["image"]["ref"]["$link"], post["author"]["did"])
+                media.append(image_media)
+        for video_media in video_medias:
+            video_media = self._download_bsky_file_as_media(video_media["ref"]["$link"], post["author"]["did"])
+            media.append(video_media)
+        return media
+
+    def _download_bsky_file_as_media(self, cid: str, did: str) -> Media:
+        """
+        Uses the Bluesky API to download a file by its `cid` and `did`.
+        """
+        # TODO: replace with self.download_from_url once that function has been cleaned-up
+        file_url = f"https://bsky.social/xrpc/com.atproto.sync.getBlob?cid={cid}&did={did}"
+        response = requests.get(file_url, stream=True)
+        response.raise_for_status()
+        ext = mimetypes.guess_extension(response.headers["Content-Type"])
+        filename = os.path.join(ArchivingContext.get_tmp_dir(), f"{cid}{ext}")
+        with open(filename, "wb") as f:
+            for chunk in response.iter_content(chunk_size=8192):
+                f.write(chunk)
+        media = Media(filename=filename)
+        media.set("src", file_url)
+        return media
+
+    def _get_post_data(self, post: dict) -> dict:
+        """
+        Extracts relevant information returned by the .getPostThread api call (excluding text/created_at): author, mentions, tags, links.
+        """
+        author = post["author"]
+        if "labels" in author and not author["labels"]: del author["labels"]
+        if "associated" in author: del author["associated"]
+
+        mentions, tags, links = [], [], []
+        facets = post.get("record", {}).get("facets", [])
+        for f in facets:
+            for feature in f["features"]:
+                if feature["$type"] == "app.bsky.richtext.facet#mention":
+                    mentions.append(feature["did"])
+                elif feature["$type"] == "app.bsky.richtext.facet#tag":
+                    tags.append(feature["tag"])
+                elif feature["$type"] == "app.bsky.richtext.facet#link":
+                    links.append(feature["uri"])
+        res = {"author": author}
+        if mentions: res["mentions"] = mentions
+        if tags: res["tags"] = tags
+        if links: res["links"] = links
+        return res
--- a/src/auto_archiver/archivers/twitter_archiver.py
+++ b/src/auto_archiver/archivers/twitter_archiver.py
@@ -1,7 +1,7 @@
-import re, requests, mimetypes, json
+import re, requests, mimetypes, json, math
+from typing import Union
 from datetime import datetime
 from loguru import logger
-from snscrape.modules.twitter import TwitterTweetScraper, Video, Gif, Photo
 from yt_dlp import YoutubeDL
 from yt_dlp.extractor.twitter import TwitterIE
 from slugify import slugify
@@ -31,7 +31,7 @@ class TwitterArchiver(Archiver):
        # expand URL if t.co and clean tracker GET params
        if 'https://t.co/' in url:
            try:
-                r = requests.get(url)
+                r = requests.get(url, timeout=30)
                logger.debug(f'Expanded url {url} to {r.url}')
                url = r.url
            except:
@@ -45,66 +45,79 @@ class TwitterArchiver(Archiver):
        can handle private/public channels
        """
        url = item.get_url()
-        # detect URLs that we definitely cannot handle
        username, tweet_id = self.get_username_tweet_id(url)
        if not username: return False

-        result = Metadata()
+        strategies = [self.download_yt_dlp, self.download_syndication]
+        for strategy in strategies:
+            logger.debug(f"Trying {strategy.__name__} for {url=}")
+            try:
+                result = strategy(item, url, tweet_id)
+                if result: return result
+            except Exception as ex:
+                logger.error(f"Failed to download {url} with {strategy.__name__}: {type(ex).__name__} occurred. args: {ex.args}")
+        
+        logger.warning(f"No free strategy worked for {url}")
+        return False
+    

-        scr = TwitterTweetScraper(tweet_id)
-        try:
-            tweet = next(scr.get_items())
-        except Exception as ex:
-            logger.warning(f"can't get tweet: {type(ex).__name__} occurred. args: {ex.args}")
-            return self.download_alternative(item, url, tweet_id)
-
-        result.set_title(tweet.content).set_content(tweet.json()).set_timestamp(tweet.date)
-        if tweet.media is None:
-            logger.debug(f'No media found, archiving tweet text only')
-            return result
-
-        for i, tweet_media in enumerate(tweet.media):
-            media = Media(filename="")
-            mimetype = ""
-            if type(tweet_media) == Video:
-                variant = max(
-                    [v for v in tweet_media.variants if v.bitrate], key=lambda v: v.bitrate)
-                media.set("src", variant.url).set("duration", tweet_media.duration)
-                mimetype = variant.contentType
-            elif type(tweet_media) == Gif:
-                variant = tweet_media.variants[0]
-                media.set("src", variant.url)
-                mimetype = variant.contentType
-            elif type(tweet_media) == Photo:
-                media.set("src", UrlUtil.twitter_best_quality_url(tweet_media.fullUrl))
-                mimetype = "image/jpeg"
-            else:
-                logger.warning(f"Could not get media URL of {tweet_media}")
-                continue
-            ext = mimetypes.guess_extension(mimetype)
-            media.filename = self.download_from_url(media.get("src"), f'{slugify(url)}_{i}{ext}')
-            result.add_media(media)
-
-        return result.success("twitter-snscrape")
-
-    def download_alternative(self, item: Metadata, url: str, tweet_id: str) -> Metadata:
+    def generate_token(self, tweet_id: str) -> str:
+        """Generates the syndication token for a tweet ID.
+        
+        Taken from https://github.com/JustAnotherArchivist/snscrape/issues/996#issuecomment-2211358215
+        And Vercel's code: https://github.com/vercel/react-tweet/blob/main/packages/react-tweet/src/api/fetch-tweet.ts#L27
        """
-        Hack alternative working again.
-        https://stackoverflow.com/a/71867055/6196010 (OUTDATED URL)
+
+        # Perform the division and multiplication by π
+        result = (int(tweet_id) / 1e15) * math.pi
+        fractional_part = result % 1
+
+        # Convert to base 36
+        base_36 = ''
+        while result >= 1:
+            base_36 = "0123456789abcdefghijklmnopqrstuvwxyz"[int(result % 36)] + base_36
+            result = math.floor(result / 36)
+
+        # Append fractional part in base 36
+        while fractional_part > 0 and len(base_36) < 11:  # Limit to avoid infinite loop
+            fractional_part *= 36
+            digit = int(fractional_part)
+            base_36 += "0123456789abcdefghijklmnopqrstuvwxyz"[digit]
+            fractional_part -= digit
+        
+        # Remove leading zeros and dots
+        return base_36.replace('0', '').replace('.', '')
+
+
+    
+    def download_syndication(self, item: Metadata, url: str, tweet_id: str) -> Union[Metadata|bool]:
+        """
+        Downloads tweets using Twitter's own embed API (Hack).
+
+        Background on method can be found at:
        https://github.com/JustAnotherArchivist/snscrape/issues/996#issuecomment-1615937362
+        https://github.com/JustAnotherArchivist/snscrape/issues/996#issuecomment-2211358215
        next to test: https://cdn.embedly.com/widgets/media.html?&schema=twitter&url=https://twitter.com/bellingcat/status/1674700676612386816
        """

-        logger.debug(f"Trying twitter hack for {url=}")
-        result = Metadata()
+        hack_url = "https://cdn.syndication.twimg.com/tweet-result"
+        params = {
+            'id': tweet_id,
+            'token': self.generate_token(tweet_id)
+        }

-        hack_url = f"https://cdn.syndication.twimg.com/tweet-result?id={tweet_id}"
-        r = requests.get(hack_url)
+        r = requests.get(hack_url, params=params, timeout=10)
        if r.status_code != 200 or r.json()=={}: 
-            logger.warning(f"Failed to get tweet information from {hack_url}, trying ytdl")
-            return self.download_ytdl(item, url, tweet_id)
+            logger.warning(f"SyndicationHack: Failed to get tweet information from {hack_url}.")
+            return False
+        
+        result = Metadata()
        tweet = r.json()

+        if tweet.get('__typename') == 'TweetTombstone':
+            logger.error(f"Failed to get tweet {tweet_id}: {tweet['tombstone']['text']['text']}")
+            return False
+
        urls = []
        for p in tweet.get("photos", []):
            urls.append(p["url"])
@@ -114,7 +127,7 @@ class TwitterArchiver(Archiver):
            v = tweet["video"]
            urls.append(self.choose_variant(v.get("variants", []))['url'])

-        logger.debug(f"Twitter hack got {urls=}")
+        logger.debug(f"Twitter hack got media {urls=}")

        for i, u in enumerate(urls):
            media = Media(filename="")
@@ -126,21 +139,30 @@ class TwitterArchiver(Archiver):

            media.filename = self.download_from_url(u, f'{slugify(url)}_{i}{ext}')
            result.add_media(media)
-
+        
        result.set_title(tweet.get("text")).set_content(json.dumps(tweet, ensure_ascii=False)).set_timestamp(datetime.strptime(tweet["created_at"], "%Y-%m-%dT%H:%M:%S.%fZ"))
-        return result.success("twitter-hack")
-    
-    def download_ytdl(self, item: Metadata, url:str, tweet_id:str) -> Metadata:
+        return result.success("twitter-syndication")
+
+    def download_yt_dlp(self, item: Metadata, url: str, tweet_id: str) -> Union[Metadata|bool]:
        downloader = YoutubeDL()
        tie = TwitterIE(downloader)
        tweet = tie._extract_status(tweet_id)
        result = Metadata()
+        try:
+            if not tweet.get("user") or not tweet.get("created_at"):
+                raise ValueError(f"Error retreiving post with id {tweet_id}. Are you sure it exists?")
+            timestamp = datetime.strptime(tweet["created_at"], "%a %b %d %H:%M:%S %z %Y")
+        except (ValueError, KeyError) as ex:
+            logger.warning(f"Unable to parse tweet: {str(ex)}\nRetreived tweet data: {tweet}")
+            return False
+                
        result\
            .set_title(tweet.get('full_text', ''))\
            .set_content(json.dumps(tweet, ensure_ascii=False))\
-            .set_timestamp(datetime.strptime(tweet["created_at"], "%a %b %d %H:%M:%S %z %Y"))
+            .set_timestamp(timestamp)
        if not tweet.get("entities", {}).get("media"):
            logger.debug('No media found, archiving tweet text only')
+            result.status = "twitter-ytdl"
            return result
        for i, tw_media in enumerate(tweet["entities"]["media"]):
            media = Media(filename="")
@@ -160,7 +182,6 @@ class TwitterArchiver(Archiver):
            media.filename = self.download_from_url(media.get("src"), f'{slugify(url)}_{i}{ext}', item)
            result.add_media(media)
        return result.success("twitter-ytdl")
-        

    def get_username_tweet_id(self, url):
        # detect URLs that we definitely cannot handle
--- a/src/auto_archiver/archivers/youtubedl_archiver.py
+++ b/src/auto_archiver/archivers/youtubedl_archiver.py
@@ -30,6 +30,8 @@ class YoutubeDLArchiver(Archiver):
            "end_means_success": {"default": True, "help": "if True, any archived content will mean a 'success', if False this archiver will not return a 'success' stage; this is useful for cases when the yt-dlp will archive a video but ignore other types of content like images or text only pages that the subsequent archivers can retrieve."},
            'allow_playlist': {"default": False, "help": "If True will also download playlists, set to False if the expectation is to download a single video."},
            "max_downloads": {"default": "inf", "help": "Use to limit the number of videos to download when a channel or long page is being extracted. 'inf' means no limit."},
+            "cookies_from_browser": {"default": None, "help": "optional browser for ytdl to extract cookies from, can be one of: brave, chrome, chromium, edge, firefox, opera, safari, vivaldi, whale"},
+            "cookie_file": {"default": None, "help": "optional cookie file to use for Youtube, see instructions here on how to export from your browser: https://github.com/yt-dlp/yt-dlp/wiki/FAQ#how-do-i-pass-cookies-to-yt-dlp"},
        }

    def download(self, item: Metadata) -> Metadata:
@@ -38,8 +40,17 @@ class YoutubeDLArchiver(Archiver):
        if item.netloc in ['facebook.com', 'www.facebook.com'] and self.facebook_cookie:
            logger.debug('Using Facebook cookie')
            yt_dlp.utils.std_headers['cookie'] = self.facebook_cookie
-
+        
        ydl_options = {'outtmpl': os.path.join(ArchivingContext.get_tmp_dir(), f'%(id)s.%(ext)s'), 'quiet': False, 'noplaylist': not self.allow_playlist , 'writesubtitles': self.subtitles, 'writeautomaticsub': self.subtitles, "live_from_start": self.live_from_start, "proxy": self.proxy, "max_downloads": self.max_downloads, "playlistend": self.max_downloads}
+
+        if item.netloc in ['youtube.com', 'www.youtube.com']:
+            if self.cookies_from_browser:
+                logger.debug(f'Extracting cookies from browser {self.cookies_from_browser} for Youtube')
+                ydl_options['cookiesfrombrowser'] = (self.cookies_from_browser,)
+            elif self.cookie_file:
+                logger.debug(f'Using cookies from file {self.cookie_file}')
+                ydl_options['cookiefile'] = self.cookie_file
+
        ydl = yt_dlp.YoutubeDL(ydl_options) # allsubtitles and subtitleslangs not working as expected, so default lang is always "en"

        try:
@@ -98,11 +109,12 @@ class YoutubeDLArchiver(Archiver):
            result.set("comments", [{
                "text": c["text"],
                "author": c["author"], 
-                "timestamp": datetime.datetime.utcfromtimestamp(c.get("timestamp")).replace(tzinfo=datetime.timezone.utc)
+                "timestamp": datetime.datetime.fromtimestamp(c.get("timestamp"), tz = datetime.timezone.utc)
            } for c in info.get("comments", [])])

        if (timestamp := info.get("timestamp")):
-            timestamp = datetime.datetime.utcfromtimestamp(timestamp).replace(tzinfo=datetime.timezone.utc).isoformat()
+            #TODO: fix deprecated timestamp, 
+            timestamp = datetime.datetime.fromtimestamp(timestamp, tz = datetime.timezone.utc).isoformat()
            result.set_timestamp(timestamp)
        if (upload_date := info.get("upload_date")):
            upload_date = datetime.datetime.strptime(upload_date, '%Y%m%d').replace(tzinfo=datetime.timezone.utc)
--- a/src/auto_archiver/enrichers/hash_enricher.py
+++ b/src/auto_archiver/enrichers/hash_enricher.py
@@ -14,9 +14,26 @@ class HashEnricher(Enricher):
    def __init__(self, config: dict) -> None:
        # without this STEP.__init__ is not called
        super().__init__(config)
-        algo_choices = self.configs()["algorithm"]["choices"]
+        algos = self.configs()["algorithm"]
+        algo_choices = algos["choices"]
+        if not getattr(self, 'algorithm', None):
+            if not config.get('algorithm'):
+                logger.warning(f"No hash algorithm selected, defaulting to {algos['default']}")
+                self.algorithm = algos["default"]
+            else:
+                self.algorithm = config["algorithm"]
+
        assert self.algorithm in algo_choices, f"Invalid hash algorithm selected, must be one of {algo_choices} (you selected {self.algorithm})."
+
+        if not getattr(self, 'chunksize', None):
+            if config.get('chunksize'):
+                self.chunksize = config["chunksize"]
+            else:
+                self.chunksize = self.configs()["chunksize"]["default"]
+
        self.chunksize = int(self.chunksize)
+        assert self.chunksize >= -1, "read length must be non-negative or -1"
+
        ArchivingContext.set("hash_enricher.algorithm", self.algorithm, keep_on_reset=True)

    @staticmethod
--- a/src/auto_archiver/enrichers/screenshot_enricher.py
+++ b/src/auto_archiver/enrichers/screenshot_enricher.py
@@ -1,5 +1,7 @@
 from loguru import logger
 import time, os
+import base64
+
 from selenium.common.exceptions import TimeoutException


@@ -18,22 +20,31 @@ class ScreenshotEnricher(Enricher):
            "timeout": {"default": 60, "help": "timeout for taking the screenshot"},
            "sleep_before_screenshot": {"default": 4, "help": "seconds to wait for the pages to load before taking screenshot"},
            "http_proxy": {"default": "", "help": "http proxy to use for the webdriver, eg http://proxy-user:password@proxy-ip:port"},
+            "save_to_pdf": {"default": False, "help": "save the page as pdf along with the screenshot. PDF saving options can be adjusted with the 'print_options' parameter"},
+            "print_options": {"default": {}, "help": "options to pass to the pdf printer"}
        }

    def enrich(self, to_enrich: Metadata) -> None:
        url = to_enrich.get_url()
+
        if UrlUtil.is_auth_wall(url):
            logger.debug(f"[SKIP] SCREENSHOT since url is behind AUTH WALL: {url=}")
            return

        logger.debug(f"Enriching screenshot for {url=}")
-        with Webdriver(self.width, self.height, self.timeout, 'facebook.com' in url, http_proxy=self.http_proxy) as driver:
+        with Webdriver(self.width, self.height, self.timeout, 'facebook.com' in url, http_proxy=self.http_proxy, print_options=self.print_options) as driver:
            try:
                driver.get(url)
                time.sleep(int(self.sleep_before_screenshot))
                screenshot_file = os.path.join(ArchivingContext.get_tmp_dir(), f"screenshot_{random_str(8)}.png")
                driver.save_screenshot(screenshot_file)
                to_enrich.add_media(Media(filename=screenshot_file), id="screenshot")
+                if self.save_to_pdf:
+                    pdf_file = os.path.join(ArchivingContext.get_tmp_dir(), f"pdf_{random_str(8)}.pdf")
+                    pdf = driver.print_page(driver.print_options)
+                    with open(pdf_file, "wb") as f:
+                        f.write(base64.b64decode(pdf))
+                    to_enrich.add_media(Media(filename=pdf_file), id="pdf")
            except TimeoutException:
                logger.info("TimeoutException loading page for screenshot")
            except Exception as e:
--- a/src/auto_archiver/enrichers/wacz_enricher.py
+++ b/src/auto_archiver/enrichers/wacz_enricher.py
@@ -34,6 +34,7 @@ class WaczArchiverEnricher(Enricher, Archiver):
            "extract_screenshot": {"default": True, "help": "If enabled the screenshot captured by browsertrix will be extracted into separate Media and appear in the html report. The .wacz file will be kept untouched."},
            "socks_proxy_host": {"default": None, "help": "SOCKS proxy host for browsertrix-crawler, use in combination with socks_proxy_port. eg: user:password@host"},
            "socks_proxy_port": {"default": None, "help": "SOCKS proxy port for browsertrix-crawler, use in combination with socks_proxy_host. eg 1234"},
+            "proxy_server": {"default": None, "help": "SOCKS server proxy URL, in development"},
        }
    
    def setup(self) -> None:
@@ -113,7 +114,10 @@ class WaczArchiverEnricher(Enricher, Archiver):
        try:
            logger.info(f"Running browsertrix-crawler: {' '.join(cmd)}")
            my_env = os.environ.copy()
-            if self.socks_proxy_host and self.socks_proxy_port:
+            if self.proxy_server:
+                logger.debug("Using PROXY_SERVER proxy for browsertrix-crawler")
+                my_env["PROXY_SERVER"] = self.proxy_server
+            elif self.socks_proxy_host and self.socks_proxy_port:
                logger.debug("Using SOCKS proxy for browsertrix-crawler")
                my_env["SOCKS_HOST"] = self.socks_proxy_host
                my_env["SOCKS_PORT"] = str(self.socks_proxy_port)
--- a/src/auto_archiver/enrichers/wayback_enricher.py
+++ b/src/auto_archiver/enrichers/wayback_enricher.py
@@ -1,3 +1,4 @@
+import json
 from loguru import logger
 import time, requests

@@ -70,11 +71,16 @@ class WaybackArchiverEnricher(Enricher, Archiver):
            return False

        # check job status
-        job_id = r.json().get('job_id')
-        if not job_id:
-            logger.error(f"Wayback failed with {r.json()}")
+        try:
+            job_id = r.json().get('job_id')
+            if not job_id:
+                logger.error(f"Wayback failed with {r.json()}")
+                return False
+        except json.decoder.JSONDecodeError as e:
+            logger.error(f"Expected a JSON with job_id from Wayback and got {r.text}")
            return False

+
        # waits at most timeout seconds until job is completed, otherwise only enriches the job_id information
        start_time = time.time()
        wayback_url = False
@@ -92,6 +98,9 @@ class WaybackArchiverEnricher(Enricher, Archiver):
            except requests.exceptions.RequestException as e:
                logger.warning(f"RequestException: fetching status for {url=} due to: {e}")
                break
+            except json.decoder.JSONDecodeError as e:
+                logger.error(f"Expected a JSON from Wayback and got {r.text} for {url=}")
+                break
            except Exception as e:
                logger.warning(f"error fetching status for {url=} due to: {e}")
            if not wayback_url:
--- a/src/auto_archiver/formatters/templates/html_template.html
+++ b/src/auto_archiver/formatters/templates/html_template.html
@@ -286,11 +286,11 @@
        // logic for enabled/disabled greyscale
        // Get references to the checkboxes and images/videos
        const safeImageViewCheckbox = document.getElementById('safe-media-view');
-        const imagesVideos = document.querySelectorAll('img, video');
+        const visualPreviews = document.querySelectorAll('img, video,embed');

        // Function to toggle grayscale effect
        function toggleGrayscale() {
-            imagesVideos.forEach(element => {
+            visualPreviews.forEach(element => {
                if (safeImageViewCheckbox.checked) {
                    // Enable grayscale effect
                    element.style.filter = 'grayscale(1)';
@@ -307,7 +307,7 @@
        safeImageViewCheckbox.addEventListener('change', toggleGrayscale);

        // Handle the hover effect using JavaScript
-        imagesVideos.forEach(element => {
+        visualPreviews.forEach(element => {
            element.addEventListener('mouseenter', () => {
                // Disable grayscale effect on hover
                element.style.filter = 'none';
--- a/src/auto_archiver/formatters/templates/macros.html
+++ b/src/auto_archiver/formatters/templates/macros.html
@@ -3,7 +3,7 @@
 {% for url in m.urls %}
 {% if url | length == 0 %}
 No URL available for {{ m.key }}.
-{% elif 'http' in url %}
+{% elif 'http://' in url or 'https://' in url or url.startswith('/') %}
 {% if 'image' in m.mimetype %}
 <div>
    <a href="{{ url }}">
@@ -32,6 +32,10 @@ No URL available for {{ m.key }}.
        Your browser does not support the video element.
    </video>
 </div>
+{% elif 'application/pdf' in m.mimetype %}
+<div>
+    <embed src="{{ url }}" width="100%" height="400px"/>
+</div>
 {% elif 'audio' in m.mimetype %}
 <div>
    <audio controls>
--- a/src/auto_archiver/utils/webdriver.py
+++ b/src/auto_archiver/utils/webdriver.py
@@ -2,18 +2,24 @@ from __future__ import annotations
 from selenium import webdriver
 from selenium.common.exceptions import TimeoutException
 from selenium.webdriver.common.proxy import Proxy, ProxyType
+from selenium.webdriver.common.print_page_options import PrintOptions
+
 from loguru import logger
 from selenium.webdriver.common.by import By
 import time


 class Webdriver:
-    def __init__(self, width: int, height: int, timeout_seconds: int, facebook_accept_cookies: bool = False, http_proxy: str = "") -> webdriver:
+    def __init__(self, width: int, height: int, timeout_seconds: int, facebook_accept_cookies: bool = False, http_proxy: str = "", print_options: dict = {}) -> webdriver:
        self.width = width
        self.height = height
        self.timeout_seconds = timeout_seconds
        self.facebook_accept_cookies = facebook_accept_cookies
        self.http_proxy = http_proxy
+        # create and set print options
+        self.print_options = PrintOptions()
+        for k, v in print_options.items():
+            setattr(self.print_options, k, v)

    def __enter__(self) -> webdriver:
        options = webdriver.FirefoxOptions()
@@ -24,6 +30,7 @@ class Webdriver:
            self.driver = webdriver.Firefox(options=options)
            self.driver.set_window_size(self.width, self.height)
            self.driver.set_page_load_timeout(self.timeout_seconds)
+            self.driver.print_options = self.print_options
        except TimeoutException as e:
            logger.error(f"failed to get new webdriver, possibly due to insufficient system resources or timeout settings: {e}")

--- a/src/auto_archiver/version.py
+++ b/src/auto_archiver/version.py
@@ -1,12 +1,12 @@
+""" Version information for the auto_archiver package.
+    TODO: This is a placeholder to replicate previous versioning.
+
+"""
+from importlib.metadata import version as get_version
+
+VERSION_SHORT = get_version("auto_archiver")

-_MAJOR = "0"
-_MINOR = "11"
-# On main and in a nightly release the patch should be one ahead of the last
-# released build.
-_PATCH = "3"
 # This is mainly for nightly builds which have the suffix ".dev$DATE". See
 # https://semver.org/#is-v123-a-semantic-version for the semantics.
 _SUFFIX = ""
-
-VERSION_SHORT = "{0}.{1}".format(_MAJOR, _MINOR)
-__version__ = "{0}.{1}.{2}{3}".format(_MAJOR, _MINOR, _PATCH, _SUFFIX)
+__version__ = f"{VERSION_SHORT}{_SUFFIX}"
--- a/tests/init.py
+++ b/tests/init.py
@@ -0,0 +1,6 @@
+import tempfile
+
+from auto_archiver.core.context import ArchivingContext
+
+ArchivingContext.reset(full_reset=True)
+ArchivingContext.set_tmp_dir(tempfile.gettempdir())
--- a/tests/archivers/init.py
+++ b/tests/archivers/init.py
--- a/tests/archivers/test_archiver_base.py
+++ b/tests/archivers/test_archiver_base.py
@@ -0,0 +1,27 @@
+import pytest
+
+from auto_archiver.core import Metadata
+from auto_archiver.core import Step
+from auto_archiver.core.metadata import Metadata
+
+class TestArchiverBase(object):
+
+    archiver_class = None
+    config = None
+
+    @pytest.fixture(autouse=True)
+    def setup_archiver(self):
+        assert self.archiver_class is not None, "self.archiver_class must be set on the subclass"
+        assert self.config is not None, "self.config must be a dict set on the subclass"
+        self.archiver = self.archiver_class(self.config)
+    
+    def assertValidResponseMetadata(self, test_response: Metadata, title: str, timestamp: str, status: str = ""):
+        assert test_response is not False
+
+        if not status:
+            assert test_response.is_success()
+        else:
+            assert status == test_response.status
+
+        assert title == test_response.get_title()
+        assert timestamp, test_response.get("timestamp")
--- a/tests/archivers/test_bluesky_archiver.py
+++ b/tests/archivers/test_bluesky_archiver.py
@@ -0,0 +1,73 @@
+import pytest
+
+from auto_archiver.archivers.bluesky_archiver import BlueskyArchiver
+from .test_archiver_base import TestArchiverBase
+
+class TestBlueskyArchiver(TestArchiverBase):
+    """Tests Bluesky Archiver
+    
+    Note that these tests will download API responses from the bluesky API, so they may be slow.
+    This is an intended feature, as we want to test to ensure the bluesky API format hasn't changed, 
+    and also test the archiver's ability to download media.
+    """
+
+    archiver_class = BlueskyArchiver
+    config = {}
+
+    @pytest.mark.download
+    def test_download_media_with_images(self):
+        # url https://bsky.app/profile/colborne.bsky.social/post/3lec2bqjc5s2y
+        post = self.archiver._get_post_from_uri("https://bsky.app/profile/colborne.bsky.social/post/3lec2bqjc5s2y")
+
+        # just make sure bsky haven't changed their format, images should be under "record/embed/media/images"
+        # there should be 2 images
+        assert "record" in post
+        assert "embed" in post["record"]
+        assert "media" in post["record"]["embed"]
+        assert "images" in post["record"]["embed"]["media"]
+        assert len(post["record"]["embed"]["media"]["images"]) == 2
+
+        # try downloading the media files
+        media = self.archiver._download_bsky_embeds(post)
+        assert len(media) == 2
+
+        # check the IDs
+        assert "bafkreiflrkfihcvwlhka5tb2opw2qog6gfvywsdzdlibveys2acozh75tq" in media[0].get('src')
+        assert "bafkreibsprmwchf7r6xcstqkdvvuj3ijw7efciw7l3y4crxr4cmynseo7u" in media[1].get('src')
+
+    @pytest.mark.download
+    def test_download_post_with_single_image(self):
+        # url https://bsky.app/profile/bellingcat.com/post/3lcxcpgt6j42l
+        post = self.archiver._get_post_from_uri("https://bsky.app/profile/bellingcat.com/post/3lcxcpgt6j42l")
+
+        # just make sure bsky haven't changed their format, images should be under "record/embed/images"
+        # there should be 1 image
+        assert "record" in post
+        assert "embed" in post["record"]
+        assert "images" in post["record"]["embed"]
+        assert len(post["record"]["embed"]["images"]) == 1
+
+        media = self.archiver._download_bsky_embeds(post)
+        assert len(media) == 1
+
+        # check the ID 
+        assert "bafkreihljdtomy4yulx4nfxuqdatlgvdg45vxdmjzzhclsd4ludk7zfma4" in media[0].get('src')
+                        
+
+    @pytest.mark.download
+    def test_download_post_with_video(self):
+        # url https://bsky.app/profile/bellingcat.com/post/3le2l4gsxlk2i
+        post = self.archiver._get_post_from_uri("https://bsky.app/profile/bellingcat.com/post/3le2l4gsxlk2i")
+
+        # just make sure bsky haven't changed their format, video should be under "record/embed/video"
+        assert "record" in post
+        assert "embed" in post["record"]
+        assert "video" in post["record"]["embed"]
+
+        media = self.archiver._download_bsky_embeds(post)
+        assert len(media) == 1
+
+        # check the ID
+        assert "bafkreiaiskn2nt5cxjnxbgcqqcrnurvkr2ni3unekn6zvhvgr5nrqg6u2q" in media[0].get('src')
+
+        
--- a/tests/archivers/test_twitter_archiver.py
+++ b/tests/archivers/test_twitter_archiver.py
@@ -0,0 +1,140 @@
+import datetime
+import pytest
+
+from auto_archiver.archivers.twitter_archiver import TwitterArchiver
+
+from .test_archiver_base import TestArchiverBase
+
+class TestTwitterArchiver(TestArchiverBase):
+
+    archiver_class = TwitterArchiver
+    config = {}
+    @pytest.mark.parametrize("url, expected", [
+        ("https://t.co/yl3oOJatFp", "https://www.bellingcat.com/category/resources/"),  # t.co URL
+        ("https://x.com/bellingcat/status/1874097816571961839", "https://x.com/bellingcat/status/1874097816571961839"), # x.com urls unchanged
+        ("https://twitter.com/bellingcat/status/1874097816571961839", "https://twitter.com/bellingcat/status/1874097816571961839"), # twitter urls unchanged
+        ("https://twitter.com/bellingcat/status/1874097816571961839?s=20&t=3d0g4ZQis7dCbSDg-mE7-w", "https://twitter.com/bellingcat/status/1874097816571961839"), # strip tracking params
+        ("https://www.bellingcat.com/category/resources/", "https://www.bellingcat.com/category/resources/"), # non-twitter/x urls unchanged
+        ("https://www.bellingcat.com/category/resources/?s=20&t=3d0g4ZQis7dCbSDg-mE7-w", "https://www.bellingcat.com/category/resources/?s=20&t=3d0g4ZQis7dCbSDg-mE7-w"), # shouldn't strip params from non-twitter/x URLs
+    ])
+    def test_sanitize_url(self, url, expected):
+        assert expected == self.archiver.sanitize_url(url)
+    
+    @pytest.mark.parametrize("url, exptected_username, exptected_tweetid", [
+        ("https://twitter.com/bellingcat/status/1874097816571961839", "bellingcat", "1874097816571961839"),
+        ("https://x.com/bellingcat/status/1874097816571961839", "bellingcat", "1874097816571961839"),
+        ("https://www.bellingcat.com/category/resources/", False, False)
+        ])
+
+    def test_get_username_tweet_id_from_url(self, url, exptected_username, exptected_tweetid):
+    
+        username, tweet_id = self.archiver.get_username_tweet_id(url)
+        assert exptected_username == username
+        assert exptected_tweetid == tweet_id
+    
+    def test_choose_variants(self):
+        # taken from the response for url https://x.com/bellingcat/status/1871552600346415571
+        variant_list = [{'content_type': 'application/x-mpegURL', 'url': 'https://video.twimg.com/ext_tw_video/1871551993677852672/pu/pl/ovWo7ux-bKROwYIC.m3u8?tag=12&v=e1b'},
+                        {'bitrate': 256000, 'content_type': 'video/mp4', 'url': 'https://video.twimg.com/ext_tw_video/1871551993677852672/pu/vid/avc1/480x270/OqZIrKV0LFswMvxS.mp4?tag=12'},
+                        {'bitrate': 832000, 'content_type': 'video/mp4', 'url': 'https://video.twimg.com/ext_tw_video/1871551993677852672/pu/vid/avc1/640x360/uiDZDSmZ8MZn9hsi.mp4?tag=12'},
+                        {'bitrate': 2176000, 'content_type': 'video/mp4', 'url': 'https://video.twimg.com/ext_tw_video/1871551993677852672/pu/vid/avc1/1280x720/6Y340Esh568WZnRZ.mp4?tag=12'}
+                        ]
+        chosen_variant = self.archiver.choose_variant(variant_list)
+        assert chosen_variant == variant_list[3]
+    
+    @pytest.mark.parametrize("tweet_id, expected_token", [
+        ("1874097816571961839", "4jjngwkifa"),
+        ("1674700676612386816", "42586mwa3uv"),
+        ("1877747914073620506", "4jv4aahw36n"),
+        ("1876710769913450647", "4jruzjz5lux"),
+        ("1346554693649113090", "39ibqxei7mo")
+        ])
+    def test_reverse_engineer_token(self, tweet_id, expected_token):
+        # see Vercel's implementation here: https://github.com/vercel/react-tweet/blob/main/packages/react-tweet/src/api/fetch-tweet.ts#L27C1-L31C2
+        # and the discussion here: https://github.com/JustAnotherArchivist/snscrape/issues/996#issuecomment-2211358215
+
+        generated_token = self.archiver.generate_token(tweet_id)
+        assert expected_token == generated_token
+
+    @pytest.mark.download
+    def test_youtube_dlp_archiver(self, make_item):
+
+        url = "https://x.com/bellingcat/status/1874097816571961839"
+        post = self.archiver.download_yt_dlp(make_item(url), url, "1874097816571961839")
+        assert post
+        self.assertValidResponseMetadata(
+            post,
+            "As 2024 comes to a close, here’s some examples of what Bellingcat investigated per month in our 10th year! 🧵",
+            datetime.datetime(2024, 12, 31, 14, 18, 33, tzinfo=datetime.timezone.utc),
+            "twitter-ytdl"
+        )
+            
+    @pytest.mark.download
+    def test_syndication_archiver(self, make_item):
+
+        url = "https://x.com/bellingcat/status/1874097816571961839"
+        post = self.archiver.download_syndication(make_item(url), url, "1874097816571961839")
+        assert post
+        self.assertValidResponseMetadata(
+            post,
+            "As 2024 comes to a close, here’s some examples of what Bellingcat investigated per month in our 10th year! 🧵",
+            datetime.datetime(2024, 12, 31, 14, 18, 33, tzinfo=datetime.timezone.utc)
+        )
+
+    @pytest.mark.download
+    def test_download_nonexistend_tweet(self, make_item):
+        # this tweet does not exist
+        url = "https://x.com/Bellingcat/status/17197025860711058"
+        response = self.archiver.download(make_item(url))
+        assert not response
+    
+    @pytest.mark.download
+    def test_download_malformed_tweetid(self, make_item):
+        # this tweet does not exist
+        url = "https://x.com/Bellingcat/status/1719702586071100058"
+        response = self.archiver.download(make_item(url))
+        assert not response
+
+    @pytest.mark.download
+    def test_download_tweet_no_media(self, make_item):
+        
+        item = make_item("https://twitter.com/MeCookieMonster/status/1617921633456640001?s=20&t=3d0g4ZQis7dCbSDg-mE7-w")
+        post = self.archiver.download(item)
+
+        self.assertValidResponseMetadata(
+            post,
+            "Onion rings are just vegetable donuts.",
+            datetime.datetime(2023, 1, 24, 16, 25, 51, tzinfo=datetime.timezone.utc),
+            "twitter-ytdl"
+        )
+    
+    @pytest.mark.download
+    def test_download_video(self, make_item):
+        url = "https://x.com/bellingcat/status/1871552600346415571"
+        post = self.archiver.download(make_item(url))
+        self.assertValidResponseMetadata(
+            post,
+            "This month's Bellingchat Premium is with @KolinaKoltai. She reveals how she investigated a platform allowing users to create AI-generated child sexual abuse material and explains why it's crucial to investigate the people behind these services https://t.co/SfBUq0hSD0 https://t.co/rIHx0WlKp8",
+            datetime.datetime(2024, 12, 24, 13, 44, 46, tzinfo=datetime.timezone.utc)
+        )
+
+    @pytest.mark.xfail(reason="Currently failing, sensitive content requires logged in users/cookies - not yet implemented")
+    @pytest.mark.download
+    @pytest.mark.parametrize("url, title, timestamp, image_hash", [
+            ("https://x.com/SozinhoRamalho/status/1876710769913450647", "ignore tweet, testing sensitivity warning nudity", datetime.datetime(2024, 12, 31, 14, 18, 33, tzinfo=datetime.timezone.utc), "image_hash"),
+            ("https://x.com/SozinhoRamalho/status/1876710875475681357", "ignore tweet, testing sensitivity warning violence", datetime.datetime(2024, 12, 31, 14, 18, 33, tzinfo=datetime.timezone.utc), "image_hash"),
+            ("https://x.com/SozinhoRamalho/status/1876711053813227618", "ignore tweet, testing sensitivity warning sensitive", datetime.datetime(2024, 12, 31, 14, 18, 33, tzinfo=datetime.timezone.utc), "image_hash"),
+            ("https://x.com/SozinhoRamalho/status/1876711141314801937", "ignore tweet, testing sensitivity warning nudity, violence, sensitivity", datetime.datetime(2024, 12, 31, 14, 18, 33, tzinfo=datetime.timezone.utc), "image_hash"),
+        ])
+    def test_download_sensitive_media(self, url, title, timestamp, image_hash, make_item):
+
+        """Download tweets with sensitive media"""
+
+        post = self.archiver.download(make_item(url))
+        self.assertValidResponseMetadata(
+            post,
+            title,
+            timestamp
+        )
+        assert len(post.media) == 1
+        assert post.media[0].hash == image_hash
--- a/tests/conftest.py
+++ b/tests/conftest.py
@@ -0,0 +1,12 @@
+import pytest
+from auto_archiver.core.metadata import Metadata
+
+@pytest.fixture
+def make_item():
+    def _make_item(url: str, **kwargs) -> Metadata:
+        item = Metadata().set_url(url)
+        for key, value in kwargs.items():
+            item.set(key, value)
+        return item
+
+    return _make_item
--- a/tests/data/testfile_1.txt
+++ b/tests/data/testfile_1.txt
@@ -0,0 +1 @@
+test1
--- a/tests/data/testfile_2.txt
+++ b/tests/data/testfile_2.txt
@@ -0,0 +1 @@
+test2
--- a/tests/databases/init.py
+++ b/tests/databases/init.py
--- a/tests/databases/test_csv_db.py
+++ b/tests/databases/test_csv_db.py
@@ -0,0 +1,22 @@
+
+from auto_archiver.databases.csv_db import CSVDb
+from auto_archiver.core import Metadata
+
+
+def test_store_item(tmp_path):
+    """Tests storing an item in the CSV database"""
+
+    temp_db = tmp_path / "temp_db.csv"
+    db = CSVDb({
+        "csv_db": {"csv_file": temp_db.as_posix()}
+        })
+
+    item = Metadata().set_url("http://example.com").set_title("Example").set_content("Example content").success("my-archiver")
+
+    db.done(item)
+
+    with open(temp_db, "r", encoding="utf-8") as f:
+        assert f.read().strip() == f"status,metadata,media\nmy-archiver: success,\"{{'_processed_at': {repr(item.get('_processed_at'))}, 'url': 'http://example.com', 'title': 'Example', 'content': 'Example content'}}\",[]"
+
+    # TODO: csv db doesn't have a fetch method - need to add it (?)
+    # assert db.fetch(item) == item
--- a/tests/enrichers/init.py
+++ b/tests/enrichers/init.py
--- a/tests/enrichers/test_hash_enricher.py
+++ b/tests/enrichers/test_hash_enricher.py
@@ -0,0 +1,55 @@
+import pytest
+
+from auto_archiver.enrichers.hash_enricher import HashEnricher
+from auto_archiver.core import Metadata, Media
+
+@pytest.mark.parametrize("algorithm, filename, expected_hash", [
+    ("SHA-256", "tests/data/testfile_1.txt", "1b4f0e9851971998e732078544c96b36c3d01cedf7caa332359d6f1d83567014"),
+    ("SHA-256", "tests/data/testfile_2.txt", "60303ae22b998861bce3b28f33eec1be758a213c86c93c076dbe9f558c11c752"),
+    ("SHA3-512", "tests/data/testfile_1.txt", "d2d8cc4f369b340130bd2b29b8b54e918b7c260c3279176da9ccaa37c96eb71735fc97568e892dc6220bf4ae0d748edb46bd75622751556393be3f482e6f794e"),
+    ("SHA3-512", "tests/data/testfile_2.txt", "e35970edaa1e0d8af7d948491b2da0450a49fd9cc1e83c5db4c6f175f9550cf341f642f6be8cfb0bfa476e4258e5088c5ad549087bf02811132ac2fa22b734c6")
+])
+def test_calculate_hash(algorithm, filename, expected_hash):
+    # test SHA-256
+    he = HashEnricher({"algorithm": algorithm, "chunksize": 1})
+    assert he.calculate_hash(filename) == expected_hash
+
+def test_default_config_values():
+    he = HashEnricher(config={})
+    assert he.algorithm == "SHA-256"
+    assert he.chunksize == 16000000
+
+def test_invalid_chunksize():
+    with pytest.raises(AssertionError):
+        he = HashEnricher({"chunksize": "-100"})
+
+def test_invalid_algorithm():
+    with pytest.raises(AssertionError):
+        HashEnricher({"algorithm": "SHA-123"})
+
+def test_config():
+    # test default config
+    c = HashEnricher.configs()
+    assert c["algorithm"]["default"] == "SHA-256"
+    assert c["chunksize"]["default"] == 16000000
+    assert c["algorithm"]["choices"] == ["SHA-256", "SHA3-512"]
+    assert c["algorithm"]["help"] == "hash algorithm to use"
+    assert c["chunksize"]["help"] == "number of bytes to use when reading files in chunks (if this value is too large you will run out of RAM), default is 16MB"
+
+def test_hash_media():
+
+    he = HashEnricher({"algorithm": "SHA-256", "chunksize": 1})
+
+    # generate metadata with two test files
+    m = Metadata().set_url("https://example.com")
+
+    # noop - the metadata has no media. Shouldn't fail
+    he.enrich(m)
+
+    m.add_media(Media("tests/data/testfile_1.txt"))
+    m.add_media(Media("tests/data/testfile_2.txt"))
+
+    he.enrich(m)
+
+    assert m.media[0].get("hash") == "SHA-256:1b4f0e9851971998e732078544c96b36c3d01cedf7caa332359d6f1d83567014"
+    assert m.media[1].get("hash") == "SHA-256:60303ae22b998861bce3b28f33eec1be758a213c86c93c076dbe9f558c11c752"
--- a/tests/formatters/init.py
+++ b/tests/formatters/init.py
--- a/tests/formatters/test_html_formatter.py
+++ b/tests/formatters/test_html_formatter.py
@@ -0,0 +1,17 @@
+from auto_archiver.core.context import ArchivingContext
+from auto_archiver.formatters.html_formatter import HtmlFormatter
+from auto_archiver.core import Metadata, Media
+
+
+def test_format():
+    formatter = HtmlFormatter({})
+    metadata = Metadata().set("content", "Hello, world!").set_url('https://example.com')
+
+    final_media = formatter.format(metadata)
+    assert isinstance(final_media, Media)
+    assert ".html" in final_media.filename
+    with open (final_media.filename, "r", encoding="utf-8") as f:
+        content = f.read()
+        assert "Hello, world!" in content
+    assert final_media.mimetype == "text/html"
+    assert "SHA-256:" in final_media.get('hash')
Author	SHA1	Message	Date
Patrick Robertson	6f10270baf	Remove unittest and switch to pytest fully	2025-01-14 16:28:39 +01:00
Patrick Robertson	cef4037ad5	Add documentation on running tests to the readme	2025-01-14 11:30:06 +01:00
Patrick Robertson	1b1af2f0b1	Revert change to twitter_archiver As per discussion at: https://github.com/bellingcat/auto-archiver/pull/165#discussion_r1905930837	2025-01-14 10:30:41 +01:00
Patrick Robertson	8f17a235f3	Switch to ubuntu-22.04 for CI tests An issue with oscrypto means it currently does not work on 24.04. Ref: https://github.com/wbond/oscrypto/issues/78#issuecomment-2565688091	2025-01-14 10:24:14 +01:00
Patrick Robertson	ab2eb3c7f5	Add dev dependencies to poetry	2025-01-13 20:42:08 +01:00
Patrick Robertson	bdfedfcf61	Merge branch 'main' into feat/unittest	2025-01-13 19:50:47 +01:00
Erin Clark	9cdaea873b	Merge pull request #164 from bellingcat/ec_add_poetry Migrate to Poetry	2025-01-13 18:49:15 +00:00
erinhmclark	84ee1b422f	Update and restrict versions of Poetry and Python.	2025-01-13 17:42:51 +00:00
Patrick Robertson	b9aea99de8	Prettify pytest output	2025-01-13 18:41:24 +01:00
Patrick Robertson	52f064908e	Add unit test badges to readme	2025-01-13 18:33:22 +01:00
Patrick Robertson	9b596e59d6	Run expensive download tests once per week, on a month at 2:35pm (time is offset from the hour to alleviate high load on Github	2025-01-13 18:33:02 +01:00
Patrick Robertson	528b78db85	Flag tombstone tweets for twitter_syndication method	2025-01-13 18:17:24 +01:00
Patrick Robertson	57eacdc24a	Merge branch 'main' into feat/unittest	2025-01-13 18:06:55 +01:00
Patrick Robertson	bbef80de4c	Add unit tests for html_formatter, csv_db	2025-01-13 17:58:10 +01:00
Patrick Robertson	930d78096a	Merge pull request #162 from bellingcat/small_issues Fix two small issues	2025-01-13 16:39:59 +01:00
Patrick Robertson	2353f9d6a5	Separate CI for download tests and core tests	2025-01-13 16:27:46 +01:00
Patrick Robertson	63973e2ce7	switch to pytest and pytest-recording	2025-01-13 16:23:20 +01:00
erinhmclark	e9a7f435a3	Add package dist directory to .gitignore	2025-01-13 13:33:23 +00:00
Patrick Robertson	e2bc84ccb9	Merge branch 'main' into feat/unittest	2025-01-13 13:15:13 +01:00
erinhmclark	72a8e76fbb	Update README.md for usage with Poetry.	2025-01-12 20:21:23 +00:00
erinhmclark	c69a5fa1c9	Refactor Dockerfile for multi-stage builds. Combining environment and runtime stages due to Poetry's dependency on source code.	2025-01-12 12:38:12 +00:00
erinhmclark	d80b4b7557	Remove snscrape and Python 3.12 restriction.	2025-01-12 12:15:56 +00:00
erinhmclark	cc490f9c10	Updated Dockerfile (not optimised yet)	2025-01-12 12:15:56 +00:00
erinhmclark	08e83eb94e	Update pyproject.toml configuration for Poetry version 2.0.0.	2025-01-12 12:15:56 +00:00
erinhmclark	dd822b8b44	Update poetry.lock	2025-01-12 12:15:56 +00:00
erinhmclark	4a63ca7753	Update PyPi workflow to read python version from pyproject.toml.	2025-01-12 12:15:56 +00:00
erinhmclark	6d5b0090d9	Pull version from pyproject.toml file/	2025-01-12 12:15:56 +00:00
erinhmclark	26abd6f7ae	Added TODO comment for adding a version restriction.	2025-01-12 12:15:56 +00:00
erinhmclark	dba8f46016	Replaced comments for python-publish.yaml workflow.	2025-01-12 12:15:56 +00:00
erinhmclark	50e8c93477	Updated workflow for python-publish.yaml to use poetry (untested), and cleanup of pipenv files.	2025-01-12 12:15:56 +00:00
erinhmclark	6da837b374	Add note to update dynamic versioning and references to version.	2025-01-12 12:15:56 +00:00
erinhmclark	660ee82c67	Update Dockerfile for poetry. Note: Review security with curl installation. Currently locked to known version, but additional checks could be added.	2025-01-12 12:15:56 +00:00
erinhmclark	5490947657	Add packaging to Poetry.	2025-01-12 12:15:56 +00:00
erinhmclark	fd9a6c26ed	Create Poetry environment. Required addition of transitive package (pyOpenSSL) and version restrictions on cryptography, boto3.	2025-01-12 12:15:56 +00:00
Patrick Robertson	3546d4ad79	Fix 'download_syndication' method for tweet archiving (now requires a token) Plus add in unit tests for token generation + download syndication	2025-01-12 12:55:00 +01:00
Patrick Robertson	c932fb7416	Improved logging when an invalid/deleted tweet is attempted to be downloaded Plus: unit tests for non-existent tweet + invalid tweet ID	2025-01-12 12:00:45 +01:00
Patrick Robertson	f29950905c	Merge branch 'main' into small_issues	2025-01-12 11:47:55 +01:00
Patrick Robertson	8e99d62c97	Merge pull request #165 from bellingcat/fix/snscrape Remove snscrape from the twitter_archiver	2025-01-09 11:06:14 +01:00
Patrick Robertson	9dc4eb35de	Switch to pytest and use vcr for request storing	2025-01-08 11:25:13 +01:00
Patrick Robertson	8c044c15f0	Add base test class for archivers with boilerplate code Plus: create test class for twitter archiver. Currently WIP	2025-01-08 10:38:56 +01:00
Patrick Robertson	ab9335bb7a	Merge branch 'main' into feat/unittest	2025-01-08 10:35:45 +01:00
Patrick Robertson	add83c9650	Remove snscrape from twitter_archiver 1. snscrape twitter downloader no longer works (ref: https://github.com/JustAnotherArchivist/snscrape/issues/1045) 2. snscrape limits python to versions <3.12	2025-01-07 19:40:19 +01:00
Miguel Sozinho Ramalho	a697f0a212	adds an unauthenticated Bluesky archiver (#160 ) * adds a TODO for next code iterations * implements bsky archiver * adds new archiver to example orchestration file * Fix downloading media for posts with multiple images (Images are stored in media/images) * Setup a basic framework for unit tests Use 'python -m unittest' from the project root to run --------- Co-authored-by: Patrick Robertson <robertson.patrick@gmail.com>	2025-01-07 10:28:07 +00:00
Patrick Robertson	bffa3a6254	Merge pull request #159 from bellingcat/print_pdf Add 'print_pdf' option to the screenshot enricher. Fixes #132	2025-01-06 18:13:38 +01:00
Miguel Sozinho Ramalho	ef471f41e1	adds better debug for wayback failures (#161 )	2025-01-06 16:49:11 +00:00
Patrick Robertson	928518cda7	Allow setting cookies for yt-dl (#158 )	2025-01-06 16:19:53 +00:00
Patrick Robertson	1bd017000e	Add Github CI test workflow	2024-12-31 15:20:33 +01:00
Patrick Robertson	33e967ce4b	Update pipfile for: - pyopenssl==24.2.1 - youtube-dlp==2024.09.27 - numpy==2.1.3 Fixes building/local installs. Also fixes #155	2024-12-31 15:20:11 +01:00
Patrick Robertson	30d423c8e6	Setup a basic framework for unit tests Use 'python -m unittest' from the project root to run	2024-12-31 14:29:52 +01:00
Patrick Robertson	0c803f15a5	Fix showing preview images in the .html file when using local storage Local storage media urls are prefixed with '/', previously only http(s) media preview src were displayed	2024-12-31 09:29:31 +01:00
Patrick Robertson	a46f9997ea	Better logging when there's a timestamp parse error	2024-12-31 09:28:08 +01:00
msramalho	83da9ae089	adds pdf preview support for html formatter	2024-12-23 18:19:26 +00:00
Patrick Robertson	663c8ad93a	Add 'print_pdf' option to the screenshot enricher. Fixes #132	2024-12-20 07:14:03 +01:00
msramalho	e49550163f	adds proxy_server option to wacz	2024-10-06 10:45:34 +06:00
msramalho	e6f5981afc	numpy version downgrade	2024-10-06 10:10:04 +06:00
msramalho	c62bf1a34d	yt-dlp version bump	2024-10-05 17:43:07 +06:00
msramalho	b166d57e61	v0.12.0 bump	2024-08-21 13:34:34 +01:00
msramalho	11c3288267	closes #146	2024-08-21 13:33:58 +01:00
msramalho	004143a58a	version bump v0.11.6	2024-07-18 11:27:39 +01:00
msramalho	686f0027c4	adds new entries to example orchestration file	2024-07-18 11:27:15 +01:00
dependabot[bot]	b03cf32c73	Bump authlib from 1.3.0 to 1.3.1 (#144 ) Bumps [authlib](https://github.com/lepture/authlib) from 1.3.0 to 1.3.1. - [Release notes](https://github.com/lepture/authlib/releases) - [Changelog](https://github.com/lepture/authlib/blob/master/docs/changelog.rst) - [Commits](https://github.com/lepture/authlib/compare/v1.3.0...v1.3.1) --- updated-dependencies: - dependency-name: authlib dependency-type: indirect ... Signed-off-by: dependabot[bot] <support@github.com> Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com>	2024-07-18 11:26:22 +01:00
msramalho	dc9e64397e	bumping yt-dlp	2024-07-18 11:23:09 +01:00
msramalho	c7bc5e2988	cleanup	2024-05-15 11:04:29 +01:00
msramalho	1e375bd740	version bump	2024-05-14 16:42:15 +01:00
Miguel Sozinho Ramalho	f8824691dd	refactors free twitter archiver strategies (#142 )	2024-05-14 16:23:33 +01:00
msramalho	012cc36609	removes deprecated datetime method	2024-05-14 15:54:50 +01:00
Miguel Sozinho Ramalho	7cfe1e39cc	#135 fix cleanup of telethon session files (#139 ) * closes #135 * version bump	2024-04-16 12:45:45 +01:00