2020-08-13 22:23:27 -04:00
|
|
|
# This is the Dockerfile for ArchiveBox, it bundles the following dependencies:
|
2022-09-12 16:34:02 -04:00
|
|
|
# python3, ArchiveBox, curl, wget, git, chromium, youtube-dl, yt-dlp, single-file
|
2019-02-28 14:04:37 -05:00
|
|
|
# Usage:
|
2022-09-11 16:11:13 -04:00
|
|
|
# git submodule update --init --recursive
|
|
|
|
# git pull --recurse-submodules
|
2020-08-13 22:23:27 -04:00
|
|
|
# docker build . -t archivebox --no-cache
|
2020-07-22 01:30:58 -04:00
|
|
|
# docker run -v "$PWD/data":/data archivebox init
|
|
|
|
# docker run -v "$PWD/data":/data archivebox add 'https://example.com'
|
2020-08-13 22:23:27 -04:00
|
|
|
# docker run -v "$PWD/data":/data -it archivebox manage createsuperuser
|
|
|
|
# docker run -v "$PWD/data":/data -p 8000:8000 archivebox server
|
2022-04-21 10:29:27 -04:00
|
|
|
# Multi-arch build:
|
|
|
|
# docker buildx create --use
|
2022-04-21 10:35:34 -04:00
|
|
|
# docker buildx build . --platform=linux/amd64,linux/arm64,linux/arm/v7 --push -t archivebox/archivebox:latest -t archivebox/archivebox:dev
|
2022-09-11 16:13:22 -04:00
|
|
|
#
|
|
|
|
# Read more about [developing
|
|
|
|
# Archivebox](https://github.com/ArchiveBox/ArchiveBox#archivebox-development).
|
2022-04-21 10:29:27 -04:00
|
|
|
|
2019-02-28 14:04:37 -05:00
|
|
|
|
2023-10-20 05:47:34 -04:00
|
|
|
FROM debian:bookworm-backports
|
2019-07-09 13:05:51 -04:00
|
|
|
|
2020-06-25 17:46:11 -04:00
|
|
|
LABEL name="archivebox" \
|
2023-10-20 05:47:34 -04:00
|
|
|
maintainer="Nick Sweeting <dockerfile@archivebox.io>" \
|
2020-08-13 22:23:27 -04:00
|
|
|
description="All-in-one personal internet archiving container" \
|
2020-11-23 02:04:39 -05:00
|
|
|
homepage="https://github.com/ArchiveBox/ArchiveBox" \
|
|
|
|
documentation="https://github.com/ArchiveBox/ArchiveBox/wiki/Docker#docker"
|
2018-10-13 22:47:30 -04:00
|
|
|
|
2023-10-20 05:47:34 -04:00
|
|
|
######### Base System Setup ####################################
|
|
|
|
|
|
|
|
# Global system-level config
|
2020-06-25 21:30:29 -04:00
|
|
|
ENV TZ=UTC \
|
2020-06-25 17:46:11 -04:00
|
|
|
LANGUAGE=en_US:en \
|
|
|
|
LC_ALL=C.UTF-8 \
|
2020-06-25 21:30:29 -04:00
|
|
|
LANG=C.UTF-8 \
|
2020-06-25 17:46:11 -04:00
|
|
|
PYTHONIOENCODING=UTF-8 \
|
|
|
|
PYTHONUNBUFFERED=1 \
|
2020-08-13 22:23:27 -04:00
|
|
|
DEBIAN_FRONTEND=noninteractive \
|
2023-10-20 05:47:34 -04:00
|
|
|
APT_KEY_DONT_WARN_ON_DANGEROUS_USAGE=1 \
|
|
|
|
npm_config_loglevel=error
|
2018-10-13 22:47:30 -04:00
|
|
|
|
2023-10-20 05:47:34 -04:00
|
|
|
# Application-level config
|
2020-08-13 22:23:27 -04:00
|
|
|
ENV CODE_DIR=/app \
|
|
|
|
DATA_DIR=/data \
|
2023-10-20 05:47:34 -04:00
|
|
|
GLOBAL_VENV=/venv \
|
|
|
|
APP_VENV=/app/.venv \
|
|
|
|
NODE_MODULES=/app/node_modules \
|
2020-08-13 22:23:27 -04:00
|
|
|
ARCHIVEBOX_USER="archivebox"
|
2020-08-03 14:19:47 -04:00
|
|
|
|
2023-10-20 05:47:34 -04:00
|
|
|
ENV PATH="$PATH:$GLOBAL_VENV/bin:$APP_VENV/bin:$NODE_MODULES/.bin"
|
|
|
|
|
|
|
|
|
2020-08-13 22:23:27 -04:00
|
|
|
# Create non-privileged user for archivebox and chrome
|
2023-10-20 07:08:38 -04:00
|
|
|
RUN echo "[*] Setting up system environment..." \
|
|
|
|
&& groupadd --system $ARCHIVEBOX_USER \
|
2023-10-20 05:47:34 -04:00
|
|
|
&& useradd --system --create-home --gid $ARCHIVEBOX_USER --groups audio,video $ARCHIVEBOX_USER \
|
|
|
|
&& mkdir -p /etc/apt/keyrings
|
2020-08-03 14:19:47 -04:00
|
|
|
|
2023-10-20 05:47:34 -04:00
|
|
|
# Install system apt dependencies (adding backports to access more recent apt updates)
|
2023-10-20 07:08:38 -04:00
|
|
|
RUN echo "[+] Installing system dependencies..." \
|
|
|
|
&& echo 'deb https://deb.debian.org/debian bullseye-backports main contrib non-free' >> /etc/apt/sources.list.d/backports.list \
|
2023-10-20 05:47:34 -04:00
|
|
|
&& apt-get update -qq \
|
|
|
|
&& apt-get install -qq -y \
|
|
|
|
apt-transport-https ca-certificates gnupg2 curl wget \
|
|
|
|
zlib1g-dev dumb-init gosu cron unzip \
|
2023-10-20 07:08:38 -04:00
|
|
|
nano iputils-ping dnsutils htop procps \
|
2023-10-20 05:47:34 -04:00
|
|
|
# 1. packaging dependencies
|
|
|
|
# 2. docker and init system dependencies
|
|
|
|
# 3. frivolous CLI helpers to make debugging failed archiving easier
|
|
|
|
&& mkdir -p /etc/apt/keyrings \
|
2020-08-13 22:23:27 -04:00
|
|
|
&& rm -rf /var/lib/apt/lists/*
|
2019-01-23 01:06:47 -05:00
|
|
|
|
2023-10-20 05:47:34 -04:00
|
|
|
|
|
|
|
######### Language Environments ####################################
|
2020-08-11 12:52:43 -04:00
|
|
|
|
2020-08-13 22:23:27 -04:00
|
|
|
# Install Node environment
|
2023-10-20 07:08:38 -04:00
|
|
|
RUN echo "[+] Installing Node environment..." \
|
2023-10-20 07:33:26 -04:00
|
|
|
&& echo 'deb [signed-by=/etc/apt/keyrings/nodesource.gpg] https://deb.nodesource.com/node_21.x nodistro main' >> /etc/apt/sources.list.d/nodejs.list \
|
2023-10-20 05:47:34 -04:00
|
|
|
&& curl -fsSL https://deb.nodesource.com/gpgkey/nodesource-repo.gpg.key | gpg --dearmor | gpg --dearmor -o /etc/apt/keyrings/nodesource.gpg \
|
2020-06-25 21:30:29 -04:00
|
|
|
&& apt-get update -qq \
|
2023-10-20 07:33:26 -04:00
|
|
|
&& apt-get install -qq -y nodejs libatomic1 \
|
2023-10-20 05:47:34 -04:00
|
|
|
&& npm i -g npm \
|
|
|
|
&& node --version \
|
|
|
|
&& npm --version
|
|
|
|
|
|
|
|
# Install Python environment
|
2023-10-20 07:08:38 -04:00
|
|
|
RUN echo "[+] Installing Python environment..." \
|
|
|
|
&& apt-get update -qq \
|
2023-10-20 05:47:34 -04:00
|
|
|
&& apt-get install -qq -y -t bookworm-backports --no-install-recommends \
|
|
|
|
python3 python3-pip python3-venv python3-setuptools python3-wheel python-dev-is-python3 \
|
2023-10-20 07:33:26 -04:00
|
|
|
python3-ldap libldap2-dev libsasl2-dev libssl-dev python3-msgpack \
|
2023-10-20 05:47:34 -04:00
|
|
|
&& rm /usr/lib/python3*/EXTERNALLY-MANAGED \
|
2023-10-20 07:08:38 -04:00
|
|
|
&& python3 -m venv --system-site-packages --symlinks $GLOBAL_VENV \
|
|
|
|
&& $GLOBAL_VENV/bin/pip install --upgrade pip pdm setuptools wheel python-ldap \
|
2020-08-13 22:23:27 -04:00
|
|
|
&& rm -rf /var/lib/apt/lists/*
|
2020-04-22 21:13:49 -04:00
|
|
|
|
2023-10-20 05:47:34 -04:00
|
|
|
######### Extractor Dependencies ##################################
|
2020-09-08 18:12:55 -04:00
|
|
|
|
2023-10-20 05:47:34 -04:00
|
|
|
# Install apt dependencies
|
2023-10-20 07:08:38 -04:00
|
|
|
RUN echo "[+] Installing extractor APT dependencies..." \
|
|
|
|
&& apt-get update -qq \
|
2023-10-20 05:47:34 -04:00
|
|
|
&& apt-get install -qq -y -t bookworm-backports --no-install-recommends \
|
|
|
|
curl wget git yt-dlp ffmpeg ripgrep \
|
|
|
|
# Packages we have also needed in the past:
|
|
|
|
# youtube-dl wget2 aria2 python3-pyxattr rtmpdump libfribidi-bin mpv \
|
|
|
|
# fontconfig fonts-ipafont-gothic fonts-wqy-zenhei fonts-thai-tlwg fonts-kacst fonts-symbola fonts-noto fonts-freefont-ttf \
|
2020-08-13 23:35:31 -04:00
|
|
|
&& rm -rf /var/lib/apt/lists/*
|
2018-10-13 22:47:30 -04:00
|
|
|
|
2023-10-20 05:47:34 -04:00
|
|
|
# Install chromium browser using playwright
|
2023-10-20 07:08:38 -04:00
|
|
|
ENV PLAYWRIGHT_BROWSERS_PATH="/browsers"
|
|
|
|
RUN echo "[+] Installing extractor Chromium dependency..." \
|
|
|
|
&& apt-get update -qq \
|
2023-10-20 05:47:34 -04:00
|
|
|
&& $GLOBAL_VENV/bin/pip install playwright \
|
|
|
|
&& $GLOBAL_VENV/bin/playwright install --with-deps chromium \
|
|
|
|
&& CHROME_BINARY="$($GLOBAL_VENV/bin/python -c 'from playwright.sync_api import sync_playwright; print(sync_playwright().start().chromium.executable_path)')" \
|
|
|
|
&& ln -s "$CHROME_BINARY" /usr/bin/chromium-browser \
|
|
|
|
&& mkdir -p "/home/${ARCHIVEBOX_USER}/.config/chromium/Crash Reports/pending/" \
|
|
|
|
&& chown -R $ARCHIVEBOX_USER "/home/${ARCHIVEBOX_USER}/.config"
|
|
|
|
|
|
|
|
# Install Node dependencies
|
|
|
|
WORKDIR "$CODE_DIR"
|
2023-10-20 07:08:38 -04:00
|
|
|
COPY --chown=root:root --chmod=755 "package.json" "package-lock.json" "$CODE_DIR/"
|
|
|
|
RUN echo "[+] Installing extractor Node dependencies..." \
|
|
|
|
&& npm ci --prefer-offline --no-audit \
|
|
|
|
&& npm version
|
2023-10-20 05:47:34 -04:00
|
|
|
|
|
|
|
######### Build Dependencies ####################################
|
2021-02-16 15:55:47 -05:00
|
|
|
|
2023-10-20 07:08:38 -04:00
|
|
|
# # Installing Python dependencies to build from source
|
|
|
|
# WORKDIR "$CODE_DIR"
|
|
|
|
# COPY --chown=root:root --chmod=755 "./pyproject.toml" "./pdm.lock" "$CODE_DIR/"
|
|
|
|
# RUN echo "[+] Installing project Python dependencies..." \
|
|
|
|
# && apt-get update -qq \
|
2023-10-20 05:47:34 -04:00
|
|
|
# && apt-get install -qq -y -t bookworm-backports --no-install-recommends \
|
|
|
|
# build-essential libssl-dev libldap2-dev libsasl2-dev \
|
2023-10-20 07:08:38 -04:00
|
|
|
# && pdm use -f $GLOBAL_VENV \
|
|
|
|
# && pdm install --fail-fast --no-lock --group :all --no-self \
|
2023-10-20 05:47:34 -04:00
|
|
|
# && pdm build \
|
|
|
|
# && apt-get purge -y \
|
|
|
|
# build-essential libssl-dev libldap2-dev libsasl2-dev \
|
|
|
|
# # these are only needed to build CPython libs, we discard after build phase to shrink layer size
|
|
|
|
# && apt-get autoremove -y \
|
|
|
|
# && rm -rf /var/lib/apt/lists/*
|
|
|
|
|
|
|
|
# Install ArchiveBox Python package from source
|
2023-10-20 07:08:38 -04:00
|
|
|
COPY --chown=root:root --chmod=755 "." "$CODE_DIR/"
|
|
|
|
RUN echo "[*] Installing ArchiveBox package from /app..." \
|
|
|
|
&& apt-get update -qq \
|
2023-10-20 05:47:34 -04:00
|
|
|
&& $GLOBAL_VENV/bin/pip install -e "$CODE_DIR"[sonic,ldap]
|
|
|
|
|
|
|
|
####################################################
|
2020-08-13 22:23:27 -04:00
|
|
|
|
|
|
|
# Setup ArchiveBox runtime config
|
2023-10-20 07:08:38 -04:00
|
|
|
WORKDIR "$DATA_DIR"
|
2020-08-10 14:15:53 -04:00
|
|
|
ENV IN_DOCKER=True \
|
2023-10-20 05:47:34 -04:00
|
|
|
WGET_BINARY="wget" \
|
|
|
|
YOUTUBEDL_BINARY="yt-dlp" \
|
2020-08-04 12:50:01 -04:00
|
|
|
CHROME_SANDBOX=False \
|
2022-04-21 10:09:17 -04:00
|
|
|
CHROME_BINARY="/usr/bin/chromium-browser" \
|
2020-08-14 12:35:35 -04:00
|
|
|
USE_SINGLEFILE=True \
|
2023-10-20 05:47:34 -04:00
|
|
|
SINGLEFILE_BINARY="$NODE_MODULES/.bin/single-file" \
|
2020-08-14 12:35:35 -04:00
|
|
|
USE_READABILITY=True \
|
2023-10-20 05:47:34 -04:00
|
|
|
READABILITY_BINARY="$NODE_MODULES/.bin/readability-extractor" \
|
2020-09-22 04:46:50 -04:00
|
|
|
USE_MERCURY=True \
|
2023-10-20 05:47:34 -04:00
|
|
|
MERCURY_BINARY="$NODE_MODULES/.bin/postlight-parser"
|
2018-10-13 22:47:30 -04:00
|
|
|
|
2020-08-13 22:23:27 -04:00
|
|
|
# Print version for nice docker finish summary
|
2020-10-27 10:11:41 -04:00
|
|
|
# RUN archivebox version
|
2023-10-20 05:47:34 -04:00
|
|
|
RUN echo "[√] Finished Docker build succesfully. Saving build summary in: /version_info.txt" \
|
|
|
|
&& uname -a | tee -a /version_info.txt \
|
|
|
|
&& env --chdir="$NODE_DIR" npm version | tee -a /version_info.txt \
|
|
|
|
&& env --chdir="$CODE_DIR" pdm info | tee -a /version_info.txt \
|
|
|
|
&& "$CODE_DIR/bin/docker_entrypoint.sh" archivebox version 2>&1 | tee -a /version_info.txt
|
|
|
|
|
|
|
|
####################################################
|
2018-10-13 22:47:30 -04:00
|
|
|
|
2020-08-13 22:23:27 -04:00
|
|
|
# Open up the interfaces to the outside world
|
2023-10-20 05:47:34 -04:00
|
|
|
VOLUME "/data"
|
2020-08-13 22:23:27 -04:00
|
|
|
EXPOSE 8000
|
2020-07-22 01:30:58 -04:00
|
|
|
|
2021-12-02 21:03:19 -05:00
|
|
|
# Optional:
|
|
|
|
# HEALTHCHECK --interval=30s --timeout=20s --retries=15 \
|
|
|
|
# CMD curl --silent 'http://localhost:8000/admin/login/' || exit 1
|
2021-02-17 18:24:38 -05:00
|
|
|
|
2020-08-10 14:15:53 -04:00
|
|
|
ENTRYPOINT ["dumb-init", "--", "/app/bin/docker_entrypoint.sh"]
|
2021-02-28 22:53:23 -05:00
|
|
|
CMD ["archivebox", "server", "--quick-init", "0.0.0.0:8000"]
|