improving insta archiver

version bump
adds tagged posts and better parsing
2026-06-10 12:18:30 +03:00 · 2024-02-23 15:37:28 +00:00 · 2024-02-23 14:08:23 +00:00 · 2024-02-23 14:08:17 +00:00 · 2024-02-23 14:08:05 +00:00 · 2024-02-22 18:05:56 +00:00
7 changed files with 111 additions and 33 deletions
--- a/src/auto_archiver/archivers/instagram_api_archiver.py
+++ b/src/auto_archiver/archivers/instagram_api_archiver.py
@@ -73,9 +73,9 @@ class InstagramAPIArchiver(Archiver):
        if type(d) == list: return [self.cleanup_dict(v) for v in d]
        if type(d) != dict: return d
        return {
-                k: self.cleanup_dict(v) if type(v) in [dict, list] else v 
+                k: clean_v
                for k, v in d.items() 
-                if v not in [0.0, 0, [], {}, "", None, "null"] and
+                if (clean_v := self.cleanup_dict(v)) not in [0.0, 0, [], {}, "", None, "null"] and
                k not in ["x", "y", "width", "height"]
        }

@@ -93,9 +93,6 @@ class InstagramAPIArchiver(Archiver):

        if self.full_profile:
            user_id = user.get("pk")
-            # download all posts
-            self.download_all_posts(result, user_id)
-
            # download all stories
            try:
                stories = self._download_stories_reusable(result, username)
@@ -104,6 +101,20 @@ class InstagramAPIArchiver(Archiver):
                result.append("errors", f"Error downloading stories for {username}")
                logger.error(f"Error downloading stories for {username}: {e}")

+            # download all posts
+            try:
+                self.download_all_posts(result, user_id)
+            except Exception as e:
+                result.append("errors", f"Error downloading posts for {username}")
+                logger.error(f"Error downloading posts for {username}: {e}")
+
+            # download all tagged
+            try:
+                self.download_all_tagged(result, user_id)
+            except Exception as e:
+                result.append("errors", f"Error downloading tagged posts for {username}")
+                logger.error(f"Error downloading tagged posts for {username}: {e}")
+
            # download all highlights
            try:
                count_highlights = 0
@@ -120,6 +131,7 @@ class InstagramAPIArchiver(Archiver):
                result.append("errors", f"Error downloading highlights for {username}")
                logger.error(f"Error downloading highlights for {username}: {e}")

+
        result.set_url(url) # reset as scrape_item modifies it
        return result.success("insta profile")

@@ -200,6 +212,28 @@ class InstagramAPIArchiver(Archiver):
                pbar.update(1)
                post_count+=1
        result.set("#posts", post_count)
+        
+    def download_all_tagged(self, result: Metadata, user_id: str):
+        next_page_id = ""
+        pbar = tqdm(desc="downloading tagged posts")
+
+        tagged_count = 0
+        while next_page_id != None:
+            resp = self.call_api(f"v2/user/tag/medias", {"user_id": user_id, "page_id": next_page_id})
+            posts = resp.get("response", {}).get("items", [])
+            if not len(posts): break
+            next_page_id = resp.get("next_page_id")
+            
+            logger.info(f"parsing {len(posts)} tagged posts, next {next_page_id=}")
+
+            for p in posts:
+                try: self.scrape_item(result, p, "tagged")
+                except Exception as e:
+                    result.append("errors", f"Error downloading tagged post {p.get('id')}")
+                    logger.error(f"Error downloading tagged post, skipping {p.get('id')}: {e}")
+                pbar.update(1)
+                tagged_count+=1
+        result.set("#tagged", tagged_count)


 ### reusable parsing utils below
@@ -217,10 +251,10 @@ class InstagramAPIArchiver(Archiver):
            if self.minimize_json_output: 
                del item["clips_metadata"]

-        if code := item.get("code"): 
-            result.set("url", f"https://www.instagram.com/p/{code}/")
+        if code := item.get("code") and not result.get("url"): 
+            result.set_url(f"https://www.instagram.com/p/{code}/")
            
-        resources = item.get("resources", [])
+        resources = item.get("resources", item.get("carousel_media", []))
        item, media, media_id = self.scrape_media(item, context)
        # if resources are present take the main media from the first resource
        if not media and len(resources):
@@ -242,7 +276,7 @@ class InstagramAPIArchiver(Archiver):
    def scrape_media(self, item: dict, context:str) -> tuple[dict, Media, str]:
        # remove unnecessary info
        if self.minimize_json_output: 
-            for k in ["image_versions", "video_versions", "video_dash_manifest"]:
+            for k in ["image_versions", "video_versions", "video_dash_manifest", "image_versions2", "video_versions2"]:
                if k in item: del item[k]
        item = self.cleanup_dict(item)

@@ -253,19 +287,24 @@ class InstagramAPIArchiver(Archiver):
            
        # retrieve video info
        best_id = item.get('id', item.get('pk'))
-        taken_at = item.get("taken_at")
+        taken_at = item.get("taken_at", item.get("taken_at_ts"))
        code = item.get("code")
+        caption_text = item.get("caption_text")
+        if "carousel_media" in item: del item["carousel_media"]
+
        if video_url := item.get("video_url"):
            filename = self.download_from_url(video_url, verbose=False)
            video_media = Media(filename=filename)
            if taken_at: video_media.set("date", taken_at)
            if code: video_media.set("url", f"https://www.instagram.com/p/{code}")
+            if caption_text: video_media.set("text", caption_text)
            video_media.set("preview", [image_media])
            video_media.set("data", [item])
            return item, video_media, f"{context or 'video'} {best_id}"
        elif image_media:
            if taken_at: image_media.set("date", taken_at)
            if code: image_media.set("url", f"https://www.instagram.com/p/{code}")
+            if caption_text: image_media.set("text", caption_text)
            image_media.set("data", [item])
            return item, image_media, f"{context or 'image'} {best_id}"
        
--- a/src/auto_archiver/archivers/youtubedl_archiver.py
+++ b/src/auto_archiver/archivers/youtubedl_archiver.py
@@ -39,7 +39,7 @@ class YoutubeDLArchiver(Archiver):
        ydl = yt_dlp.YoutubeDL(ydl_options) # allsubtitles and subtitleslangs not working as expected, so default lang is always "en"

        try:
-            # don'd download since it can be a live stream
+            # don't download since it can be a live stream
            info = ydl.extract_info(url, download=False)
            if info.get('is_live', False) and not self.livestreams:
                logger.warning("Livestream detected, skipping due to 'livestreams' configuration setting")
@@ -64,13 +64,17 @@ class YoutubeDLArchiver(Archiver):

        result = Metadata()
        result.set_title(info.get("title"))
+        if "description" in info: result.set_content(info["description"])
        for entry in entries:
            try:
                filename = ydl.prepare_filename(entry)
                if not os.path.exists(filename):
                    filename = filename.split('.')[0] + '.mkv'
-                new_media = Media(filename).set("duration", info.get("duration"))
-                
+
+                new_media = Media(filename)
+                for x in ["duration", "original_url", "fulltitle", "description", "upload_date"]:
+                    if x in entry: new_media.set(x, entry[x])
+
                # read text from subtitles if enabled
                if self.subtitles:
                    for lang, val in (info.get('requested_subtitles') or {}).items():
--- a/src/auto_archiver/core/orchestrator.py
+++ b/src/auto_archiver/core/orchestrator.py
@@ -90,7 +90,9 @@ class ArchivingOrchestrator:
        if cached_result:
            logger.debug("Found previously archived entry")
            for d in self.databases:
-                d.done(cached_result, cached=True)
+                try: d.done(cached_result, cached=True)
+                except Exception as e:
+                    logger.error(f"ERROR database {d.name}: {e}: {traceback.format_exc()}")
            return cached_result

        # 3 - call archivers until one succeeds
@@ -120,6 +122,9 @@ class ArchivingOrchestrator:
            result.status = "nothing archived"

        # signal completion to databases and archivers
-        for d in self.databases: d.done(result)
+        for d in self.databases:
+            try: d.done(result)
+            except Exception as e:
+                logger.error(f"ERROR database {d.name}: {e}: {traceback.format_exc()}")

        return result
--- a/src/auto_archiver/databases/api_db.py
+++ b/src/auto_archiver/databases/api_db.py
@@ -23,8 +23,7 @@ class AAApiDb(Database):
    def configs() -> dict:
        return {
            "api_endpoint": {"default": None, "help": "API endpoint where calls are made to"},
-            "api_secret": {"default": None, "help": "API Basic authentication secret [deprecating soon]"},
-            "api_token": {"default": None, "help": "API Bearer token, to be preferred over secret (Basic auth) going forward"},
+            "api_token": {"default": None, "help": "API Bearer token."},
            "public": {"default": False, "help": "whether the URL should be publicly available via the API"},
            "author_id": {"default": None, "help": "which email to assign as author"},
            "group_id": {"default": None, "help": "which group of users have access to the archive in case public=false as author"},
@@ -59,7 +58,7 @@ class AAApiDb(Database):
        logger.debug(f"saving archive of {item.get_url()} to the AA API.")

        payload = {'result': item.to_json(), 'public': self.public, 'author_id': self.author_id, 'group_id': self.group_id, 'tags': list(self.tags)}
-        headers = {"Authorization": f"Bearer {self.api_secret}"}
+        headers = {"Authorization": f"Bearer {self.api_token}"}
        response = requests.post(os.path.join(self.api_endpoint, "submit-archive"), json=payload, headers=headers)

        if response.status_code == 200:
--- a/src/auto_archiver/enrichers/wacz_enricher.py
+++ b/src/auto_archiver/enrichers/wacz_enricher.py
@@ -35,6 +35,22 @@ class WaczArchiverEnricher(Enricher, Archiver):
            "socks_proxy_host": {"default": None, "help": "SOCKS proxy host for browsertrix-crawler, use in combination with socks_proxy_port. eg: user:password@host"},
            "socks_proxy_port": {"default": None, "help": "SOCKS proxy port for browsertrix-crawler, use in combination with socks_proxy_host. eg 1234"},
        }
+    
+    def setup(self) -> None:
+        self.use_docker = os.environ.get('WACZ_ENABLE_DOCKER') or not os.environ.get('RUNNING_IN_DOCKER')
+        self.docker_in_docker = os.environ.get('WACZ_ENABLE_DOCKER') and os.environ.get('RUNNING_IN_DOCKER')
+
+        self.cwd_dind = f"/crawls/crawls{random_str(8)}"
+        self.browsertrix_home_host = os.environ.get('BROWSERTRIX_HOME_HOST')
+        self.browsertrix_home_container = os.environ.get('BROWSERTRIX_HOME_CONTAINER') or self.browsertrix_home_host
+        # create crawls folder if not exists, so it can be safely removed in cleanup
+        if self.docker_in_docker:
+            os.makedirs(self.cwd_dind, exist_ok=True)
+
+    def cleanup(self) -> None:
+        if self.docker_in_docker:
+            logger.debug(f"Removing {self.cwd_dind=}")
+            shutil.rmtree(self.cwd_dind, ignore_errors=True)

    def download(self, item: Metadata) -> Metadata:
        # this new Metadata object is required to avoid duplication
@@ -51,8 +67,8 @@ class WaczArchiverEnricher(Enricher, Archiver):
        url = to_enrich.get_url()

        collection = random_str(8)
-        browsertrix_home_host = os.environ.get('BROWSERTRIX_HOME_HOST') or os.path.abspath(ArchivingContext.get_tmp_dir())
-        browsertrix_home_container = os.environ.get('BROWSERTRIX_HOME_CONTAINER') or browsertrix_home_host
+        browsertrix_home_host = self.browsertrix_home_host or os.path.abspath(ArchivingContext.get_tmp_dir())
+        browsertrix_home_container = self.browsertrix_home_container or browsertrix_home_host

        cmd = [
            "crawl",
@@ -67,11 +83,12 @@ class WaczArchiverEnricher(Enricher, Archiver):
            "--behaviors", "autoscroll,autoplay,autofetch,siteSpecific",
            "--behaviorTimeout", str(self.timeout),
            "--timeout", str(self.timeout)]
+        
+        if self.docker_in_docker:
+            cmd.extend(["--cwd", self.cwd_dind])

        # call docker if explicitly enabled or we are running on the host (not in docker)
-        use_docker = os.environ.get('WACZ_ENABLE_DOCKER') or not os.environ.get('RUNNING_IN_DOCKER')
-
-        if use_docker:
+        if self.use_docker:
            logger.debug(f"generating WACZ in Docker for {url=}")
            logger.debug(f"{browsertrix_home_host=} {browsertrix_home_container=}")
            if self.docker_commands:
@@ -103,7 +120,10 @@ class WaczArchiverEnricher(Enricher, Archiver):
            logger.error(f"WACZ generation failed: {e}")
            return False

-        if use_docker:
+        
+        if self.docker_in_docker:
+            wacz_fn = os.path.join(self.cwd_dind, "collections", collection, f"{collection}.wacz")
+        elif self.use_docker:
            wacz_fn = os.path.join(browsertrix_home_container, "collections", collection, f"{collection}.wacz")
        else:
            wacz_fn = os.path.join("collections", collection, f"{collection}.wacz")
@@ -116,7 +136,9 @@ class WaczArchiverEnricher(Enricher, Archiver):
        if self.extract_media or self.extract_screenshot:
            self.extract_media_from_wacz(to_enrich, wacz_fn)

-        if use_docker:
+        if self.docker_in_docker:
+            jsonl_fn = os.path.join(self.cwd_dind, "collections", collection, "pages", "pages.jsonl")
+        elif self.use_docker:
            jsonl_fn = os.path.join(browsertrix_home_container, "collections", collection, "pages", "pages.jsonl")
        else:
            jsonl_fn = os.path.join("collections", collection, "pages", "pages.jsonl")
--- a/src/auto_archiver/formatters/templates/html_template.html
+++ b/src/auto_archiver/formatters/templates/html_template.html
@@ -177,14 +177,23 @@
    }

    async function run() {
-        await PreviewCertificates();
-        await PreviewText();
-        await enableCopyLogic();
-        await enableCollapsibleLogic();
-        await setupSafeView();
+        let setupFunctions = [
+            previewCertificates,
+            previewText,
+            enableCopyLogic,
+            enableCollapsibleLogic,
+            setupSafeView
+        ];
+        setupFunctions.forEach(async f => {
+            try {
+                await f();
+            } catch (e) {
+                console.error(`Error in ${f.name}: ${e}`);
+            }
+        });
    }

-    async function PreviewCertificates() {
+    async function previewCertificates() {
        await Promise.all(
            Array.from(document.querySelectorAll(".pem-certificate")).map(async el => {
                let certificate = await (await fetch(el.getAttribute("pem"))).text();
@@ -202,7 +211,7 @@
        console.log("certificate preview done");
    }

-    async function PreviewText() {
+    async function previewText() {
        await Promise.all(
            Array.from(document.querySelectorAll(".text-preview")).map(async el => {
                let textContent = await (await fetch(el.getAttribute("url"))).text();
--- a/src/auto_archiver/version.py
+++ b/src/auto_archiver/version.py
@@ -3,7 +3,7 @@ _MAJOR = "0"
 _MINOR = "9"
 # On main and in a nightly release the patch should be one ahead of the last
 # released build.
-_PATCH = "0"
+_PATCH = "9"
 # This is mainly for nightly builds which have the suffix ".dev$DATE". See
 # https://semver.org/#is-v123-a-semantic-version for the semantics.
 _SUFFIX = ""
Author	SHA1	Message	Date
msramalho	70075a1e5e	improving insta archiver	2024-02-23 15:37:28 +00:00
msramalho	5b9bc4919a	version bump	2024-02-23 14:08:23 +00:00
msramalho	f0158ffd9c	adds tagged posts and better parsing	2024-02-23 14:08:17 +00:00
msramalho	bfb35a43a9	adds more details from yt-dlp	2024-02-23 14:08:05 +00:00
msramalho	ef5b39c4f1	dind exception	2024-02-22 18:05:56 +00:00
msramalho	24ceafcb64	missing forward slash	2024-02-22 17:47:13 +00:00
msramalho	9fd4bb56a8	new attempt at dind wacz	2024-02-22 17:24:27 +00:00
msramalho	5324d562ba	cleanup wacz patch	2024-02-21 18:14:30 +00:00
msramalho	5bf0a0206d	version update	2024-02-21 17:26:07 +00:00
msramalho	4941823565	fix growing volume size in wacz_enricher	2024-02-21 17:25:55 +00:00
msramalho	27310c2911	fixes issue with api requests	2024-02-21 12:25:05 +00:00
msramalho	eb973ba42d	v0.9.1 fixes to bad parsing in ssl certificates	2024-02-20 19:31:19 +00:00