mirror of
https://github.com/bellingcat/auto-archiver.git
synced 2026-06-12 21:28:29 +03:00
Compare commits
11 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
53494c961e | ||
|
|
f7839a99cc | ||
|
|
7a2119e6e9 | ||
|
|
3ae25e51e7 | ||
|
|
9584193d69 | ||
|
|
0dd45d90f1 | ||
|
|
edcb2da74a | ||
|
|
17d9bf694f | ||
|
|
368395ffa8 | ||
|
|
21d7d2e16c | ||
|
|
0bbb4c9b08 |
@@ -4,7 +4,6 @@ ENV RUNNING_IN_DOCKER=1
|
|||||||
|
|
||||||
WORKDIR /app
|
WORKDIR /app
|
||||||
|
|
||||||
# TODO: use custom ffmpeg builds instead of apt-get install
|
|
||||||
RUN pip install --upgrade pip && \
|
RUN pip install --upgrade pip && \
|
||||||
pip install pipenv && \
|
pip install pipenv && \
|
||||||
add-apt-repository ppa:mozillateam/ppa && \
|
add-apt-repository ppa:mozillateam/ppa && \
|
||||||
@@ -18,7 +17,6 @@ RUN pip install --upgrade pip && \
|
|||||||
rm geckodriver-v*
|
rm geckodriver-v*
|
||||||
|
|
||||||
|
|
||||||
# TODO: avoid copying unnecessary files, including .git
|
|
||||||
COPY Pipfile* ./
|
COPY Pipfile* ./
|
||||||
# install from pipenv, with browsertrix-only requirements
|
# install from pipenv, with browsertrix-only requirements
|
||||||
RUN pipenv install && \
|
RUN pipenv install && \
|
||||||
@@ -27,12 +25,7 @@ RUN pipenv install && \
|
|||||||
# doing this at the end helps during development, builds are quick
|
# doing this at the end helps during development, builds are quick
|
||||||
COPY ./src/ .
|
COPY ./src/ .
|
||||||
|
|
||||||
# TODO: figure out how to make volumes not be root, does it depend on host or dockerfile?
|
|
||||||
# RUN useradd --system --groups sudo --shell /bin/bash archiver && chown -R archiver:sudo .
|
|
||||||
# USER archiver
|
|
||||||
|
|
||||||
|
|
||||||
ENTRYPOINT ["pipenv", "run", "python3", "-m", "auto_archiver"]
|
ENTRYPOINT ["pipenv", "run", "python3", "-m", "auto_archiver"]
|
||||||
|
|
||||||
# should be executed with 2 volumes (3 if local_storage is used)
|
# should be executed with 2 volumes (3 if local_storage is used)
|
||||||
# docker run --rm -v $PWD/secrets:/app/secrets -v $PWD/local_archive:/app/local_archive aa pipenv run python3 -m auto_archiver --config secrets/orchestration.yaml
|
# docker run --rm -v $PWD/secrets:/app/secrets -v $PWD/local_archive:/app/local_archive aa pipenv run python3 -m auto_archiver --config secrets/orchestration.yaml
|
||||||
|
|||||||
@@ -25,7 +25,7 @@ class YoutubeDLArchiver(Archiver):
|
|||||||
logger.debug('Using Facebook cookie')
|
logger.debug('Using Facebook cookie')
|
||||||
yt_dlp.utils.std_headers['cookie'] = self.facebook_cookie
|
yt_dlp.utils.std_headers['cookie'] = self.facebook_cookie
|
||||||
|
|
||||||
ydl = yt_dlp.YoutubeDL({'outtmpl': os.path.join(ArchivingContext.get_tmp_dir(), f'%(id)s.%(ext)s'), 'quiet': False})
|
ydl = yt_dlp.YoutubeDL({'outtmpl': os.path.join(ArchivingContext.get_tmp_dir(), f'%(id)s.%(ext)s'), 'quiet': False, 'noplaylist': True})
|
||||||
|
|
||||||
try:
|
try:
|
||||||
# don'd download since it can be a live stream
|
# don'd download since it can be a live stream
|
||||||
|
|||||||
@@ -27,6 +27,7 @@ class WaczArchiverEnricher(Enricher, Archiver):
|
|||||||
def configs() -> dict:
|
def configs() -> dict:
|
||||||
return {
|
return {
|
||||||
"profile": {"default": None, "help": "browsertrix-profile (for profile generation see https://github.com/webrecorder/browsertrix-crawler#creating-and-using-browser-profiles)."},
|
"profile": {"default": None, "help": "browsertrix-profile (for profile generation see https://github.com/webrecorder/browsertrix-crawler#creating-and-using-browser-profiles)."},
|
||||||
|
"docker_commands": {"default": None, "help":"if a custom docker invocation is needed"},
|
||||||
"timeout": {"default": 120, "help": "timeout for WACZ generation in seconds"},
|
"timeout": {"default": 120, "help": "timeout for WACZ generation in seconds"},
|
||||||
"extract_media": {"default": True, "help": "If enabled all the images/videos/audio present in the WACZ archive will be extracted into separate Media. The .wacz file will be kept untouched."}
|
"extract_media": {"default": True, "help": "If enabled all the images/videos/audio present in the WACZ archive will be extracted into separate Media. The .wacz file will be kept untouched."}
|
||||||
}
|
}
|
||||||
@@ -46,51 +47,44 @@ class WaczArchiverEnricher(Enricher, Archiver):
|
|||||||
url = to_enrich.get_url()
|
url = to_enrich.get_url()
|
||||||
|
|
||||||
collection = str(uuid.uuid4())[0:8]
|
collection = str(uuid.uuid4())[0:8]
|
||||||
browsertrix_home = os.path.abspath(ArchivingContext.get_tmp_dir())
|
browsertrix_home_host = os.environ.get('BROWSERTRIX_HOME_HOST') or os.path.abspath(ArchivingContext.get_tmp_dir())
|
||||||
|
browsertrix_home_container = os.environ.get('BROWSERTRIX_HOME_CONTAINER') or browsertrix_home_host
|
||||||
|
|
||||||
if os.getenv('RUNNING_IN_DOCKER'):
|
cmd = [
|
||||||
|
"crawl",
|
||||||
|
"--url", url,
|
||||||
|
"--scopeType", "page",
|
||||||
|
"--generateWACZ",
|
||||||
|
"--text",
|
||||||
|
"--screenshot", "fullPage",
|
||||||
|
"--collection", collection,
|
||||||
|
"--id", collection,
|
||||||
|
"--saveState", "never",
|
||||||
|
"--behaviors", "autoscroll,autoplay,autofetch,siteSpecific",
|
||||||
|
"--behaviorTimeout", str(self.timeout),
|
||||||
|
"--timeout", str(self.timeout)]
|
||||||
|
|
||||||
|
# call docker if explicitly enabled or we are running on the host (not in docker)
|
||||||
|
use_docker = os.environ.get('WACZ_ENABLE_DOCKER') or not os.environ.get('RUNNING_IN_DOCKER')
|
||||||
|
|
||||||
|
if use_docker:
|
||||||
|
logger.debug(f"generating WACZ in Docker for {url=}")
|
||||||
|
logger.debug(f"{browsertrix_home_host=} {browsertrix_home_container=}")
|
||||||
|
if not self.docker_commands:
|
||||||
|
self.docker_commands = ["docker", "run", "--rm", "-v", f"{browsertrix_home_host}:/crawls/", "webrecorder/browsertrix-crawler"]
|
||||||
|
cmd = self.docker_commands + cmd
|
||||||
|
|
||||||
|
if self.profile:
|
||||||
|
profile_fn = os.path.join(browsertrix_home_container, "profile.tar.gz")
|
||||||
|
logger.debug(f"copying {self.profile} to {profile_fn}")
|
||||||
|
shutil.copyfile(self.profile, profile_fn)
|
||||||
|
cmd.extend(["--profile", os.path.join("/crawls", "profile.tar.gz")])
|
||||||
|
|
||||||
|
else:
|
||||||
logger.debug(f"generating WACZ without Docker for {url=}")
|
logger.debug(f"generating WACZ without Docker for {url=}")
|
||||||
|
|
||||||
cmd = [
|
|
||||||
"crawl",
|
|
||||||
"--url", url,
|
|
||||||
"--scopeType", "page",
|
|
||||||
"--generateWACZ",
|
|
||||||
"--text",
|
|
||||||
"--screenshot", "fullPage",
|
|
||||||
"--collection", collection,
|
|
||||||
"--id", collection,
|
|
||||||
"--saveState", "never",
|
|
||||||
"--behaviors", "autoscroll,autoplay,autofetch,siteSpecific",
|
|
||||||
"--behaviorTimeout", str(self.timeout),
|
|
||||||
"--timeout", str(self.timeout)]
|
|
||||||
|
|
||||||
if self.profile:
|
if self.profile:
|
||||||
cmd.extend(["--profile", os.path.join("/app", str(self.profile))])
|
cmd.extend(["--profile", os.path.join("/app", str(self.profile))])
|
||||||
else:
|
|
||||||
logger.debug(f"generating WACZ in Docker for {url=}")
|
|
||||||
|
|
||||||
cmd = [
|
|
||||||
"docker", "run",
|
|
||||||
"--rm", # delete container once it has completed running
|
|
||||||
"-v", f"{browsertrix_home}:/crawls/",
|
|
||||||
# "-it", # this leads to "the input device is not a TTY"
|
|
||||||
"webrecorder/browsertrix-crawler", "crawl",
|
|
||||||
"--url", url,
|
|
||||||
"--scopeType", "page",
|
|
||||||
"--generateWACZ",
|
|
||||||
"--text",
|
|
||||||
"--screenshot", "fullPage",
|
|
||||||
"--collection", collection,
|
|
||||||
"--behaviors", "autoscroll,autoplay,autofetch,siteSpecific",
|
|
||||||
"--behaviorTimeout", str(self.timeout),
|
|
||||||
"--timeout", str(self.timeout)
|
|
||||||
]
|
|
||||||
|
|
||||||
if self.profile:
|
|
||||||
profile_fn = os.path.join(browsertrix_home, "profile.tar.gz")
|
|
||||||
shutil.copyfile(self.profile, profile_fn)
|
|
||||||
cmd.extend(["--profile", os.path.join("/crawls", "profile.tar.gz")])
|
|
||||||
|
|
||||||
try:
|
try:
|
||||||
logger.info(f"Running browsertrix-crawler: {' '.join(cmd)}")
|
logger.info(f"Running browsertrix-crawler: {' '.join(cmd)}")
|
||||||
@@ -99,10 +93,10 @@ class WaczArchiverEnricher(Enricher, Archiver):
|
|||||||
logger.error(f"WACZ generation failed: {e}")
|
logger.error(f"WACZ generation failed: {e}")
|
||||||
return False
|
return False
|
||||||
|
|
||||||
if os.getenv('RUNNING_IN_DOCKER'):
|
if use_docker:
|
||||||
filename = os.path.join("collections", collection, f"{collection}.wacz")
|
filename = os.path.join(browsertrix_home_container, "collections", collection, f"{collection}.wacz")
|
||||||
else:
|
else:
|
||||||
filename = os.path.join(browsertrix_home, "collections", collection, f"{collection}.wacz")
|
filename = os.path.join("collections", collection, f"{collection}.wacz")
|
||||||
|
|
||||||
if not os.path.exists(filename):
|
if not os.path.exists(filename):
|
||||||
logger.warning(f"Unable to locate and upload WACZ {filename=}")
|
logger.warning(f"Unable to locate and upload WACZ {filename=}")
|
||||||
|
|||||||
@@ -3,7 +3,7 @@ _MAJOR = "0"
|
|||||||
_MINOR = "6"
|
_MINOR = "6"
|
||||||
# On main and in a nightly release the patch should be one ahead of the last
|
# On main and in a nightly release the patch should be one ahead of the last
|
||||||
# released build.
|
# released build.
|
||||||
_PATCH = "6"
|
_PATCH = "10"
|
||||||
# This is mainly for nightly builds which have the suffix ".dev$DATE". See
|
# This is mainly for nightly builds which have the suffix ".dev$DATE". See
|
||||||
# https://semver.org/#is-v123-a-semantic-version for the semantics.
|
# https://semver.org/#is-v123-a-semantic-version for the semantics.
|
||||||
_SUFFIX = ""
|
_SUFFIX = ""
|
||||||
|
|||||||
Reference in New Issue
Block a user