Bump version to v0.6.0 for release

minor improvements
feat: WACZ enricher can now be probed for media, and used as an archiver OR enricher
2026-06-10 04:08:28 +03:00 · 2023-07-27 15:42:29 +01:00 · 2023-07-27 15:42:23 +01:00 · 2023-07-27 15:42:10 +01:00 · 2023-07-26 16:13:14 +01:00 · 2023-07-26 16:12:56 +01:00
13 changed files with 680 additions and 601 deletions
--- a/1
+++ b/1
@@ -36,6 +36,7 @@ uwsgi = "*"
 requests = {extras = ["socks"], version = "*"}
 # wacz = "==0.4.8"
 numpy = "*"
+warcio = "*"

 [requires]
 python_version = "3.10"
--- a/Pipfile.lock
+++ b/Pipfile.lock
--- a/example.orchestration.yaml
+++ b/example.orchestration.yaml
@@ -12,13 +12,14 @@ steps:
    # - tiktok_archiver
    - youtubedl_archiver
    # - wayback_archiver_enricher
+    # - wacz_archiver_enricher
  enrichers:
    - hash_enricher
    # - screenshot_enricher
    # - thumbnail_enricher
    # - wayback_archiver_enricher
-    # - wacz_enricher
-    # - pdq_hash_enricher
+    # - wacz_archiver_enricher
+    # - pdq_hash_enricher # if you want to calculate hashes for thumbnails, include this after thumbnail_enricher
  formatter: html_formatter # defaults to mute_formatter
  storages:
    - local_storage
@@ -95,7 +96,7 @@ configurations:
    secret: "wayback secret"
  hash_enricher:
    algorithm: "SHA3-512" # can also be SHA-256
-  wacz_enricher:
+  wacz_archiver_enricher:
    profile: secrets/profile.tar.gz
  local_storage:
    save_to: "./local_archive"
--- a/src/auto_archiver/archivers/telegram_archiver.py
+++ b/src/auto_archiver/archivers/telegram_archiver.py
@@ -49,7 +49,6 @@ class TelegramArchiver(Archiver):
        if video is None:
            logger.warning("could not find video")
            image_tags = s.find_all(class_="tgme_widget_message_photo_wrap")
-            logger.info(image_tags)

            image_urls = []
            for im in image_tags:
--- a/src/auto_archiver/archivers/twitter_archiver.py
+++ b/src/auto_archiver/archivers/twitter_archiver.py
@@ -6,6 +6,7 @@ from slugify import slugify

 from . import Archiver
 from ..core import Metadata, Media
+from ..utils import UrlUtil


 class TwitterArchiver(Archiver):
@@ -77,7 +78,7 @@ class TwitterArchiver(Archiver):
                media.set("src", variant.url)
                mimetype = variant.contentType
            elif type(tweet_media) == Photo:
-                media.set("src", tweet_media.fullUrl.replace('name=large', 'name=orig'))
+                media.set("src", tweet_media.fullUrl.replace('name=large', 'name=orig').replace('name=small', 'name=orig'))
                mimetype = "image/jpeg"
            else:
                logger.warning(f"Could not get media URL of {tweet_media}")
@@ -95,21 +96,7 @@ class TwitterArchiver(Archiver):
        https://github.com/JustAnotherArchivist/snscrape/issues/996#issuecomment-1615937362
        next to test: https://cdn.embedly.com/widgets/media.html?&schema=twitter&url=https://twitter.com/bellingcat/status/1674700676612386816
        """
-        headers = {
-            "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:109.0) Gecko/20100101 Firefox/114.0",
-            "Accept": "*/*",
-            "Accept-Language": "en-US,en;q=0.5",
-            "Accept-Encoding": "gzip, deflate, br",
-            "Origin": "https://platform.twitter.com",
-            "Connection": "keep-alive",
-            "Referer": "https://platform.twitter.com/",
-            "Sec-Fetch-Dest": "empty",
-            "Sec-Fetch-Mode": "cors",
-            "Sec-Fetch-Site": "cross-site",
-            "Pragma": "no-cache",
-            "Cache-Control": "no-cache",
-            "TE": "trailers"
-        }
+
        logger.debug(f"Trying twitter hack for {url=}")
        result = Metadata()

@@ -133,7 +120,7 @@ class TwitterArchiver(Archiver):
            media = Media(filename="")
            media.set("src", u)
            ext = ""
-            if (mtype := mimetypes.guess_type(u)[0]):
+            if (mtype := mimetypes.guess_type(UrlUtil.remove_get_parameters(u))[0]):
                ext = mimetypes.guess_extension(mtype)

            media.filename = self.download_from_url(u, f'{slugify(url)}_{i}{ext}', item)
--- a/src/auto_archiver/core/orchestrator.py
+++ b/src/auto_archiver/core/orchestrator.py
@@ -109,6 +109,8 @@ class ArchivingOrchestrator:
        # looks for Media in result.media and also result.media[x].properties (as list or dict values)
        result.store()

+        #TODO: remove any duplicate media, if hash is available
+
        # 6 - format and store formatted if needed
        # enrichers typically need access to already stored URLs etc
        if (final_media := self.formatter.format(result)):
--- a/src/auto_archiver/enrichers/init.py
+++ b/src/auto_archiver/enrichers/init.py
@@ -3,6 +3,6 @@ from .screenshot_enricher import ScreenshotEnricher
 from .wayback_enricher import WaybackArchiverEnricher
 from .hash_enricher import HashEnricher
 from .thumbnail_enricher import ThumbnailEnricher
-from .wacz_enricher import WaczEnricher
+from .wacz_enricher import WaczArchiverEnricher
 from .whisper_enricher import WhisperEnricher
 from .pdq_hash_enricher import PdqHashEnricher
--- a/src/auto_archiver/enrichers/pdq_hash_enricher.py
+++ b/src/auto_archiver/enrichers/pdq_hash_enricher.py
@@ -1,6 +1,7 @@
+import traceback
 import pdqhash
 import numpy as np
-from PIL import Image
+from PIL import Image, UnidentifiedImageError
 from loguru import logger

 from . import Enricher
@@ -32,11 +33,15 @@ class PdqHashEnricher(Enricher):
                        media.set("pdq_hash", hd)    

    def calculate_pdq_hash(self, filename):
-        # returns a hexadecimal string with the perceptual hash for the given filename 
-        with Image.open(filename) as img:
-            # convert the image to RGB
-            image_rgb = np.array(img.convert("RGB"))
-            # compute the 256-bit PDQ hash (we do not store the quality score)
-            hash_array, _ = pdqhash.compute(image_rgb)
-            hash = "".join(str(b) for b in hash_array)
-            return hex(int(hash, 2))[2:]
+        # returns a hexadecimal string with the perceptual hash for the given filename
+        try:
+            with Image.open(filename) as img:
+                # convert the image to RGB
+                image_rgb = np.array(img.convert("RGB"))
+                # compute the 256-bit PDQ hash (we do not store the quality score)
+                hash_array, _ = pdqhash.compute(image_rgb)
+                hash = "".join(str(b) for b in hash_array)
+                return hex(int(hash, 2))[2:]
+        except UnidentifiedImageError as e:
+            logger.error(f"Image {filename=} is likely corrupted or in unsupported format {e}: {traceback.format_exc()}")
+        return ""
--- a/src/auto_archiver/enrichers/wacz_enricher.py
+++ b/src/auto_archiver/enrichers/wacz_enricher.py
@@ -1,16 +1,23 @@
+import mimetypes
 import os, shutil, subprocess, uuid
+from zipfile import ZipFile
 from loguru import logger
+from warcio.archiveiterator import ArchiveIterator

 from ..core import Media, Metadata, ArchivingContext
 from . import Enricher
+from ..archivers import Archiver
 from ..utils import UrlUtil


-class WaczEnricher(Enricher):
+class WaczArchiverEnricher(Enricher, Archiver):
    """
-    Submits the current URL to the webarchive and returns a job_id or completed archive
+    Uses https://github.com/webrecorder/browsertrix-crawler to generate a .WACZ archive of the URL
+    If used with [profiles](https://github.com/webrecorder/browsertrix-crawler#creating-and-using-browser-profiles)
+    it can become quite powerful for archiving private content.
+    When used as an archiver it will extract the media from the .WACZ archive so it can be enriched.
    """
-    name = "wacz_enricher"
+    name = "wacz_archiver_enricher"

    def __init__(self, config: dict) -> None:
        # without this STEP.__init__ is not called
@@ -20,16 +27,28 @@ class WaczEnricher(Enricher):
    def configs() -> dict:
        return {
            "profile": {"default": None, "help": "browsertrix-profile (for profile generation see https://github.com/webrecorder/browsertrix-crawler#creating-and-using-browser-profiles)."},
-            "timeout": {"default": 90, "help": "timeout for WACZ generation in seconds"},
-            "ignore_auth_wall": {"default": True, "help": "skip URL if it is behind authentication wall, set to False if you have browsertrix profile configured for private content."},
+            "timeout": {"default": 120, "help": "timeout for WACZ generation in seconds"},
+            "extract_media": {"default": True, "help": "If enabled all the images/videos/audio present in the WACZ archive will be extracted into separate Media. The .wacz file will be kept untouched."}
        }

+    def download(self, item: Metadata) -> Metadata:
+        # this new Metadata object is required to avoid duplication
+        result = Metadata()
+        result.merge(item)
+        if self.enrich(result):
+            return result.success("wacz")
+
    def enrich(self, to_enrich: Metadata) -> bool:
+        if to_enrich.get_media_by_id("browsertrix"):
+            logger.info(f"WACZ enricher had already been executed: {to_enrich.get_media_by_id('browsertrix')}")
+            return True
+
        url = to_enrich.get_url()
-        
+        logger.warning(f"ENRICHING WACZ for {url=}")
+
        collection = str(uuid.uuid4())[0:8]
        browsertrix_home = os.path.abspath(ArchivingContext.get_tmp_dir())
-        
+
        if os.getenv('RUNNING_IN_DOCKER'):
            logger.debug(f"generating WACZ without Docker for {url=}")

@@ -45,12 +64,12 @@ class WaczEnricher(Enricher):
                "--behaviors", "autoscroll,autoplay,autofetch,siteSpecific",
                "--behaviorTimeout", str(self.timeout),
                "--timeout", str(self.timeout)]
-            
+
            if self.profile:
                cmd.extend(["--profile", os.path.join("/app", str(self.profile))])
        else:
            logger.debug(f"generating WACZ in Docker for {url=}")
-            
+
            cmd = [
                "docker", "run",
                "--rm",  # delete container once it has completed running
@@ -79,15 +98,65 @@ class WaczEnricher(Enricher):
            logger.error(f"WACZ generation failed: {e}")
            return False

-        
-
        if os.getenv('RUNNING_IN_DOCKER'):
            filename = os.path.join("collections", collection, f"{collection}.wacz")
        else:
            filename = os.path.join(browsertrix_home, "collections", collection, f"{collection}.wacz")
-        
+
        if not os.path.exists(filename):
            logger.warning(f"Unable to locate and upload WACZ  {filename=}")
            return False

        to_enrich.add_media(Media(filename), "browsertrix")
+        if self.extract_media:
+            self.extract_media_from_wacz(to_enrich, filename)
+        return True
+
+    def extract_media_from_wacz(self, to_enrich: Metadata, wacz_filename: str) -> None:
+        """
+        Receives a .wacz archive, and extracts all relevant media from it, adding them to to_enrich.
+        """
+        logger.info(f"WACZ extract_media flag is set, extracting media from {wacz_filename=}")
+
+        # unzipping the .wacz
+        tmp_dir = ArchivingContext.get_tmp_dir()
+        unzipped_dir = os.path.join(tmp_dir, "unzipped")
+        with ZipFile(wacz_filename, 'r') as z_obj:
+            z_obj.extractall(path=unzipped_dir)
+
+        # if warc is split into multiple gzip chunks, merge those
+        warc_dir = os.path.join(unzipped_dir, "archive")
+        warc_filename = os.path.join(tmp_dir, "merged.warc")
+        with open(warc_filename, 'wb') as outfile:
+            for filename in sorted(os.listdir(warc_dir)):
+                if filename.endswith('.gz'):
+                    chunk_file = os.path.join(warc_dir, filename)
+                    with open(chunk_file, 'rb') as infile:
+                        shutil.copyfileobj(infile, outfile)
+
+        # get media out of .warc
+        counter = 0
+        with open(warc_filename, 'rb') as warc_stream:
+            for record in ArchiveIterator(warc_stream):
+                # only include fetched resources
+                if record.rec_type != 'response': continue
+                record_url = record.rec_headers.get_header('WARC-Target-URI')
+                if not UrlUtil.is_relevant_url(record_url):
+                    logger.debug(f"Skipping irrelevant URL {record_url} but it's still present in the WACZ.")
+                    continue
+
+                # filter by media mimetypes
+                content_type = record.http_headers.get("Content-Type")
+                if not content_type: continue
+                if not any(x in content_type for x in ["video", "image", "audio"]): continue
+
+                # create local file and add media
+                ext = mimetypes.guess_extension(content_type)
+                fn = os.path.join(tmp_dir, f"warc-file-{counter}{ext}")
+                with open(fn, "wb") as outf: outf.write(record.raw_stream.read())
+                m = Media(filename=fn)
+                m.set("src", record_url)
+                # TODO URLUTIL to ignore known-recurring media like favicons, profile pictures, etc.
+                to_enrich.add_media(m, f"browsertrix-media-{counter}")
+                counter += 1
+        logger.info(f"WACZ extract_media finished, found {counter} relevant media file(s)")
--- a/src/auto_archiver/enrichers/wayback_enricher.py
+++ b/src/auto_archiver/enrichers/wayback_enricher.py
@@ -28,6 +28,7 @@ class WaybackArchiverEnricher(Enricher, Archiver):
        }

    def download(self, item: Metadata) -> Metadata:
+        # this new Metadata object is required to avoid duplication
        result = Metadata()
        result.merge(item)
        if self.enrich(result):
--- a/src/auto_archiver/utils/misc.py
+++ b/src/auto_archiver/utils/misc.py
@@ -20,7 +20,6 @@ def expand_url(url):
            logger.error(f'Failed to expand url {url}')
    return url

-
 def getattr_or(o: object, prop: str, default=None):
    try:
        res = getattr(o, prop)
--- a/src/auto_archiver/utils/url.py
+++ b/src/auto_archiver/utils/url.py
@@ -1,14 +1,16 @@
 import re
+from urllib.parse import urlparse, urlunparse
+

 class UrlUtil:
    telegram_private = re.compile(r"https:\/\/t\.me(\/c)\/(.+)\/(\d+)")
    is_istagram = re.compile(r"https:\/\/www\.instagram\.com")

    @staticmethod
-    def clean(url): return url
+    def clean(url: str) -> str: return url

    @staticmethod
-    def is_auth_wall(url):
+    def is_auth_wall(url: str) -> bool:
        """
        checks if URL is behind an authentication wall meaning steps like wayback, wacz, ... may not work
        """
@@ -17,3 +19,28 @@ class UrlUtil:

        return False

+    @staticmethod
+    def remove_get_parameters(url: str) -> str:
+        # http://example.com/file.mp4?t=1 -> http://example.com/file.mp4
+        # useful for mimetypes to work
+        parsed_url = urlparse(url)
+        new_url = urlunparse(parsed_url._replace(query=''))
+        return new_url
+
+    @staticmethod
+    def is_relevant_url(url: str) -> bool:
+        """
+        Detect if a detected media URL is recurring and therefore irrelevant to a specific archive. Useful, for example, for the enumeration of the media files in WARC files which include profile pictures, favicons, etc.
+        """
+        clean_url = UrlUtil.remove_get_parameters(url)
+
+        # favicons
+        if "favicon" in url: return False
+        # ifnore icons
+        if clean_url.endswith(".ico"): return False
+        # ignore SVGs
+        if UrlUtil.remove_get_parameters(url).endswith(".svg"): return False
+
+        # twitter profile pictures
+        if "twimg.com/profile_images" in url: return False
+        return True
--- a/src/auto_archiver/version.py
+++ b/src/auto_archiver/version.py
@@ -1,9 +1,9 @@

 _MAJOR = "0"
-_MINOR = "5"
+_MINOR = "6"
 # On main and in a nightly release the patch should be one ahead of the last
 # released build.
-_PATCH = "27"
+_PATCH = "0"
 # This is mainly for nightly builds which have the suffix ".dev$DATE". See
 # https://semver.org/#is-v123-a-semantic-version for the semantics.
 _SUFFIX = ""
Author	SHA1	Message	Date
msramalho	1e66a2c905	Bump version to v0.6.0 for release	2023-07-27 15:42:29 +01:00
msramalho	e8f44b652e	minor improvements	2023-07-27 15:42:23 +01:00
msramalho	dd034da844	feat: WACZ enricher can now be probed for media, and used as an archiver OR enricher	2023-07-27 15:42:10 +01:00
msramalho	65e3c99483	Bump version to v0.5.28 for release	2023-07-26 16:13:14 +01:00
msramalho	888ad8f004	fix: twitter hack videos extension detection	2023-07-26 16:12:56 +01:00
msramalho	086a9e6c84	fix: remove unnecessary log	2023-07-11 12:17:15 +01:00