One example per way of selecting what to fetch (urls-file, --profile, --all, --post-url), plus the routine/stories/full-sweep pacing choices, so the flags don't have to be reverse-engineered from the arg list. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_011qAds5qr7nZRq5R4yAuxUk
971 lines
40 KiB
Python
Executable File
971 lines
40 KiB
Python
Executable File
#!/usr/bin/env python3
|
|
"""
|
|
Fetch Instagram profiles into the archive layout using gallery-dl.
|
|
|
|
The CLI replacement for the JDownloader2 workflow. See docs/gallery-dl.md for
|
|
the measurements behind every choice here — especially the safety model, which
|
|
is the reason this script exists in this shape rather than a simpler one.
|
|
|
|
The fetch host needs no copy of the archive. It stages locally and rsyncs
|
|
afterwards; what it already holds is learned from a *file listing* alone
|
|
(`--index`), which the viewer's own API serves.
|
|
|
|
Usage:
|
|
./scripts/gdl-sync.py --index https://instaarchive.ergosteur.com \\
|
|
--staging /var/tmp/gdl --publish user@host:/path/to/archives \\
|
|
--urls-file artms_account_links.txt --dry-run
|
|
|
|
# ...then swap --dry-run for --execute. --index also accepts a local path,
|
|
# and --profile / --all work instead of --urls-file.
|
|
|
|
Always --dry-run first: it prints the plan, and the publish step it reports is
|
|
the one that would touch the archive.
|
|
|
|
Run it from the host whose public IP matches the browser the cookie came from;
|
|
using the cookie from elsewhere is what session-hijack detection looks for.
|
|
|
|
See `--help` for every flag, with worked examples for each way of selecting
|
|
what to fetch.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
EXAMPLES = """\
|
|
examples:
|
|
|
|
Routine incremental sync of the tracked profiles (what gdl-cron.sh runs) --
|
|
--abort 50 stops enumerating each profile once it reaches content already
|
|
held, so a run that has seeded once costs ~40-60 requests, not a full walk:
|
|
|
|
./scripts/gdl-sync.py --index https://instaarchive.ergosteur.com \\
|
|
--staging /var/tmp/gdl --publish user@host:/path/to/archives \\
|
|
--urls-file artms_account_links.txt --archive-db /var/tmp/gdl.db \\
|
|
--abort 50 --dry-run
|
|
# ...then swap --dry-run for --execute once the plan looks right.
|
|
|
|
Cheapest possible run -- stories only, the one surface that expires in 24h
|
|
and cannot be backfilled, so it is worth doing often:
|
|
|
|
./scripts/gdl-sync.py --index https://instaarchive.ergosteur.com \\
|
|
--staging /var/tmp/gdl --publish user@host:/path/to/archives \\
|
|
--urls-file artms_account_links.txt --only stories --execute
|
|
|
|
One profile, by name, without a urls-file:
|
|
|
|
./scripts/gdl-sync.py --index /path/to/archives --staging /var/tmp/gdl \\
|
|
--publish /path/to/archives --profile some_account --execute
|
|
|
|
Every profile the archive already knows about (no urls-file, no --profile):
|
|
|
|
./scripts/gdl-sync.py --index /path/to/archives --staging /var/tmp/gdl \\
|
|
--publish /path/to/archives --all --execute
|
|
|
|
An arbitrary single post or reel from an account NOT otherwise tracked --
|
|
e.g. a link someone shared. Filed under its owner like any other post; no
|
|
--index needed, since there is no profile list to plan against:
|
|
|
|
./scripts/gdl-sync.py --staging /var/tmp/gdl --publish user@host:/path \\
|
|
--post-url https://www.instagram.com/p/SHORTCODE/ --execute
|
|
|
|
Full sweep -- no --abort, walks every profile to the end. The only run that
|
|
notices a carousel edited after it was archived, and by far the most
|
|
expensive thing here (~420 requests for six profiles). Read docs/gallery-dl.md
|
|
and TOOLING.md before running this one:
|
|
|
|
./scripts/gdl-sync.py --index https://instaarchive.ergosteur.com \\
|
|
--staging /var/tmp/gdl --publish user@host:/path/to/archives \\
|
|
--urls-file artms_account_links.txt --execute
|
|
|
|
Hand-paced caution after a scraping warning (roughly double the defaults;
|
|
see docs/gallery-dl.md for where these numbers come from):
|
|
|
|
./scripts/gdl-sync.py ... --sleep-request 12 20 --sleep 5 10 --rate 500K
|
|
|
|
Always --dry-run first (the default): it prints the plan and the rsync
|
|
command that would publish, without spending a single Instagram request.
|
|
"""
|
|
|
|
import argparse
|
|
import datetime as dt
|
|
import json
|
|
import os
|
|
import re
|
|
import shutil
|
|
import subprocess
|
|
import sys
|
|
from dataclasses import dataclass, field
|
|
from pathlib import Path
|
|
|
|
# --------------------------------------------------------------------------
|
|
# Archive layout
|
|
# --------------------------------------------------------------------------
|
|
|
|
# Mirrors src/lib/archive-grouping.ts. Instagram usernames cannot contain
|
|
# spaces, which is what makes the username separable from a highlight title.
|
|
RE_HIGHLIGHT = re.compile(r"^story highlights - ([^ ]+) - (.+)$")
|
|
RE_STORIES = re.compile(r"^story - ([^ ]+)$")
|
|
RE_REELS = re.compile(r"^([^ ]+) - reels$")
|
|
|
|
DATE_FMT = "{date:Olocal/%Y-%m-%d}"
|
|
"""Local-time date. JD2 stamped US Eastern, NOT UTC (0/212 mismatches vs 19 for
|
|
UTC). `Olocal` is DST-aware per timestamp. The trailing separator must be
|
|
omitted or it lands in the strftime format and sanitises to an underscore."""
|
|
|
|
POST_STEM = DATE_FMT + "_{username} - {post_shortcode}"
|
|
ITEM_STEM = DATE_FMT + "_{username} - {shortcode}"
|
|
|
|
|
|
@dataclass
|
|
class Source:
|
|
"""One gallery-dl invocation: a URL fetched into a specific directory."""
|
|
|
|
kind: str # posts | reels | stories | highlights
|
|
url: str
|
|
directory: str # relative to the archives root
|
|
subcategory: str # gallery-dl config key
|
|
title: str | None = None # highlight title, when known
|
|
|
|
|
|
@dataclass
|
|
class Profile:
|
|
user: str
|
|
existing: dict[str, str] = field(default_factory=dict) # kind -> dirname
|
|
|
|
def sources(self, kinds: set[str]) -> list[Source]:
|
|
u = self.user
|
|
base = f"https://www.instagram.com/{u}"
|
|
all_sources = [
|
|
Source("posts", f"{base}/posts/", u, "posts"),
|
|
Source("reels", f"{base}/reels/", f"{u} - reels", "reels"),
|
|
# Stories expire after 24h, so these can only ever be captured
|
|
# live. There is no backfill and no re-fetch -- which is why they
|
|
# are the one surface worth visiting daily.
|
|
Source("stories", f"https://www.instagram.com/stories/{u}/",
|
|
f"story - {u}", "stories"),
|
|
# Highlight directories embed the title, which gallery-dl only
|
|
# learns mid-extraction -- so this one source fans out into many
|
|
# directories and is handled with a directory format string.
|
|
Source("highlights", f"{base}/highlights", "", "highlights"),
|
|
]
|
|
return [s for s in all_sources if s.kind in kinds]
|
|
|
|
|
|
def scan_archives(root: Path) -> dict[str, Profile]:
|
|
"""Group existing directories into profiles, as the server does."""
|
|
profiles: dict[str, Profile] = {}
|
|
|
|
def get(user: str) -> Profile:
|
|
return profiles.setdefault(user, Profile(user))
|
|
|
|
for entry in sorted(os.listdir(root)):
|
|
if not (root / entry).is_dir() or entry.startswith("."):
|
|
continue
|
|
if m := RE_HIGHLIGHT.match(entry):
|
|
get(m.group(1)).existing.setdefault("highlights", entry)
|
|
elif m := RE_STORIES.match(entry):
|
|
get(m.group(1)).existing["stories"] = entry
|
|
elif m := RE_REELS.match(entry):
|
|
get(m.group(1)).existing["reels"] = entry
|
|
else:
|
|
get(entry).existing["posts"] = entry
|
|
return profiles
|
|
|
|
|
|
RE_PROFILE_URL = re.compile(
|
|
r"^(?:https?://)?(?:www\.)?instagram\.com/(?P<user>[^/?#\s]+)/?", re.I)
|
|
|
|
# Path segments that are Instagram features, not profiles. A line like
|
|
# ".../p/ABC123/" names a post, and treating "p" as a username would silently
|
|
# sync nothing under a nonsense directory.
|
|
RESERVED_SEGMENTS = {
|
|
"p", "reel", "reels", "stories", "explore", "accounts", "direct",
|
|
"tv", "s", "invites", "challenge", "about", "developer",
|
|
}
|
|
|
|
|
|
def read_urls_file(path: Path) -> list[str]:
|
|
"""
|
|
Read profile URLs (or bare usernames) from a file, one per line.
|
|
|
|
Written for hand-maintained lists: blank lines are skipped, `#` starts a
|
|
comment, and either a full URL or a bare username works. Order is kept and
|
|
duplicates dropped, so a list can be appended to without care.
|
|
"""
|
|
users: list[str] = []
|
|
seen: set[str] = set()
|
|
|
|
for lineno, raw in enumerate(path.read_text().splitlines(), 1):
|
|
line = raw.split("#", 1)[0].strip()
|
|
if not line:
|
|
continue
|
|
|
|
m = RE_PROFILE_URL.match(line)
|
|
user = m.group("user") if m else line.strip("/")
|
|
|
|
if not user or "/" in user or " " in user:
|
|
print(f"{path}:{lineno}: cannot read a username from {raw.strip()!r}",
|
|
file=sys.stderr)
|
|
continue
|
|
if user.lower() in RESERVED_SEGMENTS:
|
|
print(f"{path}:{lineno}: {user!r} is an Instagram path, not a "
|
|
f"profile — skipping", file=sys.stderr)
|
|
continue
|
|
if user in seen:
|
|
continue
|
|
|
|
seen.add(user)
|
|
users.append(user)
|
|
|
|
return users
|
|
|
|
|
|
class ArchiveIndex:
|
|
"""
|
|
What the archive already holds, as filenames only.
|
|
|
|
Deliberately never reads file *contents*, so the fetch host does not need a
|
|
copy of the archive — it can stage locally and rsync afterwards. Backed
|
|
either by a local directory or by the viewer's own API, which already
|
|
serves exactly this listing and is the cheaper option when the archive
|
|
lives on network storage (a full walk there took ~52s).
|
|
"""
|
|
|
|
def __init__(self, source: str):
|
|
self.remote = source.startswith(("http://", "https://"))
|
|
self.source = source.rstrip("/") if self.remote else None
|
|
self.root = None if self.remote else Path(source)
|
|
if self.root and not self.root.is_dir():
|
|
raise SystemExit(f"archive index not found: {source}")
|
|
self._cache: dict[str, list[str]] = {}
|
|
|
|
def _get(self, path: str):
|
|
from urllib.request import urlopen
|
|
with urlopen(f"{self.source}{path}", timeout=60) as resp:
|
|
return json.load(resp)
|
|
|
|
def profiles(self) -> set[str]:
|
|
if self.remote:
|
|
return {a["name"] for a in self._get("/api/archives")}
|
|
return set(scan_archives(self.root))
|
|
|
|
def listing(self, user: str) -> list[str]:
|
|
"""Every filename belonging to a profile, across all its sidecars."""
|
|
if user in self._cache:
|
|
return self._cache[user]
|
|
|
|
names: list[str] = []
|
|
if self.remote:
|
|
try:
|
|
data = self._get(f"/api/archives/{user}/files")
|
|
except Exception:
|
|
data = []
|
|
files = data if isinstance(data, list) else data.get("files", [])
|
|
names = [f["path"] for f in files]
|
|
else:
|
|
prof = scan_archives(self.root).get(user)
|
|
for dirname in (prof.existing.values() if prof else ()):
|
|
d = self.root / dirname
|
|
if d.is_dir():
|
|
names += [f"{dirname}/{n}" for n in os.listdir(d)]
|
|
|
|
self._cache[user] = names
|
|
return names
|
|
|
|
|
|
# --------------------------------------------------------------------------
|
|
# gallery-dl configuration
|
|
# --------------------------------------------------------------------------
|
|
|
|
def build_config(rate: str, sleep_request: list[float],
|
|
sleep: list[float], abort: int = 0) -> dict:
|
|
"""
|
|
The config is generated rather than checked in so the safety-critical
|
|
options cannot drift out of sync with the docs.
|
|
|
|
`api: rest` is the single most important line in this file. The graphql
|
|
backend issues one request PER POST for every video and carousel, which is
|
|
the pattern that got this account banned once already.
|
|
"""
|
|
caption_pp = {
|
|
"name": "metadata",
|
|
"event": "post",
|
|
"mode": "custom",
|
|
"content-format": "{description}",
|
|
"extension": "txt",
|
|
# JD2 wrote no .txt when the caption was empty; "empty": false (the
|
|
# default) reproduces that.
|
|
}
|
|
meta_pp = {
|
|
"name": "metadata",
|
|
"event": "post",
|
|
"mode": "json",
|
|
# `include`, NOT `fields` -- `fields` applies to mode:custom and
|
|
# silently does nothing here, dumping audio_user blobs that contain
|
|
# unrelated users' profile picture URLs.
|
|
"include": [
|
|
"post_shortcode", "post_id", "type", "date", "post_date",
|
|
"username", "fullname", "owner_id", "description", "count",
|
|
"likes", "post_url", "sidecar_shortcode",
|
|
],
|
|
}
|
|
|
|
def post_like(stem: str) -> dict:
|
|
"""Naming for surfaces whose unit is a post (posts, reels)."""
|
|
skip: dict = {}
|
|
if abort:
|
|
# Stop enumerating once `abort` consecutive files are already in
|
|
# the skip-archive. The listing pass -- not the downloading -- is
|
|
# what costs `instagram.com` requests, and it otherwise walks the
|
|
# whole profile every run to find three new posts.
|
|
#
|
|
# Safe here only because the REST listing is strictly
|
|
# reverse-chronological: the web grid hoists pinned posts to the
|
|
# front, but this endpoint does not (measured 2026-08-20), so old
|
|
# posts never appear before new ones.
|
|
#
|
|
# Counted in FILES, not posts, so it must clear the largest
|
|
# already-held carousel -- 22 media for one real post in this
|
|
# archive. It also means edited carousels (test case 15) stop
|
|
# being noticed, so a full sweep is still worth running
|
|
# occasionally.
|
|
skip["skip"] = f"abort:{abort}"
|
|
return {
|
|
**skip,
|
|
# `sidecar_shortcode` is set only for carousels, so it is the
|
|
# carousel discriminator. First matching condition wins.
|
|
"filename": {
|
|
"sidecar_shortcode and count >= 10":
|
|
stem + " - {num:02}.{extension}",
|
|
"sidecar_shortcode":
|
|
stem + " - {num}.{extension}",
|
|
"":
|
|
stem + ".{extension}",
|
|
},
|
|
"postprocessors": [
|
|
{**caption_pp, "filename": stem + ".txt"},
|
|
{**meta_pp, "filename": stem + ".json"},
|
|
],
|
|
}
|
|
|
|
def item_like(stem: str) -> dict:
|
|
"""
|
|
Naming for surfaces whose unit is an item inside a reel (stories,
|
|
highlights). `{shortcode}` is per item; `{post_shortcode}` is the
|
|
reel's id and is shared by every item in it.
|
|
|
|
The media filename uses the per-item shortcode, but the sidecar cannot:
|
|
it runs at `event: post`, where the kwdict describes the *reel* and has
|
|
no `shortcode` at all -- which silently formatted as the literal
|
|
"None", producing one "<date>_<user> - None.json" per reel. It is keyed
|
|
by `post_shortcode` instead, and is genuinely reel-level data (the
|
|
reel's own date and item count); per-item dates live in the media
|
|
filenames, which is the more precise source anyway.
|
|
"""
|
|
return {
|
|
"filename": stem + ".{extension}",
|
|
"postprocessors": [
|
|
{**meta_pp,
|
|
"filename": DATE_FMT + "_{username} - {post_shortcode}.json"},
|
|
],
|
|
}
|
|
|
|
return {
|
|
"extractor": {
|
|
"base-directory": ".",
|
|
"instagram": {
|
|
"api": "rest", # never "graphql" -- see docstring
|
|
"sleep-request": sleep_request,
|
|
"sleep": sleep,
|
|
# The CDN does rate-limit: a first run at 3M/1-3s drew
|
|
# '429 Too Many Requests' from scontent-*.cdninstagram.com and
|
|
# lost two videos. Back off hard rather than retry fast.
|
|
"sleep-429": 120.0,
|
|
"retries": 8,
|
|
"videos": True,
|
|
"include": "", # never "all"; sources are explicit
|
|
# Directory is forced per-invocation with -D, because a reels
|
|
# tab returns collab reels owned by OTHER accounts and
|
|
# {username} would scatter them into the wrong profile.
|
|
"directory": [],
|
|
"posts": post_like(POST_STEM),
|
|
"reels": post_like(POST_STEM),
|
|
# Ad hoc single-post fetches (--post-url) go through gallery-dl's
|
|
# own post/reel extractor instead of a profile listing, so the
|
|
# owning account is never known ahead of time -- only mid-
|
|
# extraction, same reasoning as highlights below.
|
|
"post": {**post_like(POST_STEM), "directory": ["{username}"]},
|
|
"reel": {**post_like(POST_STEM), "directory": ["{username}"]},
|
|
"stories": item_like(ITEM_STEM),
|
|
"highlights": {
|
|
**item_like(ITEM_STEM),
|
|
# The only surface that must derive its own directory,
|
|
# since the title is not known until extraction.
|
|
"directory": ["story highlights - {username} - {highlight_title}"],
|
|
},
|
|
},
|
|
},
|
|
# `retries` here is the CDN-side counterpart to sleep-429 above.
|
|
"downloader": {"http": {"rate": rate, "retries": 8}},
|
|
"output": {"mode": "null"},
|
|
}
|
|
|
|
|
|
# --------------------------------------------------------------------------
|
|
# Planning and execution
|
|
# --------------------------------------------------------------------------
|
|
|
|
def gdl_command(src: Source, staging: Path, config: Path, cookies: str,
|
|
archive_db: Path | None) -> list[str]:
|
|
cmd = [
|
|
"gallery-dl",
|
|
"--config", str(config),
|
|
"--cookies-from-browser", cookies,
|
|
]
|
|
if archive_db:
|
|
# Without a seeded skip-archive, staging is empty and every file is
|
|
# re-downloaded; see seed_archive_db.
|
|
cmd += ["--download-archive", str(archive_db)]
|
|
# Forced destination -- never `{username}` -- because a reels tab returns
|
|
# collab reels owned by other accounts, which would otherwise be filed
|
|
# under the wrong profile. Highlights and ad hoc single-post fetches are
|
|
# the exception: their directory is only known mid-extraction (a title, or
|
|
# the post's owner), so the config formats it instead.
|
|
dest = (staging if src.subcategory in ("highlights", "post", "reel")
|
|
else staging / src.directory)
|
|
cmd += ["--destination", str(dest)]
|
|
cmd.append(src.url)
|
|
return cmd
|
|
|
|
|
|
# gallery-dl keys its skip-archive on `archive_prefix + archive_fmt`, which for
|
|
# this extractor is the literal "instagram" followed by the per-media numeric
|
|
# pk (`instagram.py:25`, `job.py:713-719`). Verified against a real run: a
|
|
# 3-image carousel produced 3 rows, one per item.
|
|
ARCHIVE_KEY = "instagram{}".format
|
|
ARCHIVE_SCHEMA = "CREATE TABLE IF NOT EXISTS archive (entry TEXT PRIMARY KEY)"
|
|
|
|
RE_ARCHIVED = re.compile(
|
|
r"^(\d{4}-\d{2}-\d{2})_(.+?) - ([A-Za-z0-9_-]+?)(?: - (\d+))?\.(\w+)$")
|
|
NON_MEDIA = {"txt", "json"}
|
|
|
|
|
|
def index_existing(listing: list[str]) -> set[tuple[str, int]]:
|
|
"""
|
|
Reduce a flat list of filenames to the (shortcode, index) pairs already
|
|
held. Only names matter — never the bytes — which is what lets the sync run
|
|
on a host that has no copy of the archive.
|
|
"""
|
|
have: set[tuple[str, int]] = set()
|
|
for name in listing:
|
|
m = RE_ARCHIVED.match(name.rsplit("/", 1)[-1])
|
|
if not m or m.group(5).lower() in NON_MEDIA:
|
|
continue
|
|
# An absent index means a single-media post, which is index 1 — the
|
|
# same normalisation the viewer's EXPORT_RE applies.
|
|
have.add((m.group(3), int(m.group(4) or 1)))
|
|
return have
|
|
|
|
|
|
def live_key(item: dict, kind: str) -> tuple[str, int]:
|
|
"""
|
|
The (shortcode, index) a live item *would* be filed under, mirroring the
|
|
filename template exactly.
|
|
|
|
The two surfaces disagree about which shortcode identifies a file, and
|
|
getting this wrong silently seeds almost nothing:
|
|
|
|
posts/reels filed under {post_shortcode} — for a carousel, each
|
|
child item ALSO has its own `shortcode`, which is not
|
|
what appears in the filename.
|
|
stories/highlights filed under the per-item {shortcode}, because
|
|
`post_shortcode` there is the containing reel's id and
|
|
is shared by every item in it.
|
|
"""
|
|
if kind in ("stories", "highlights"):
|
|
return (item.get("shortcode"), 1)
|
|
return (item.get("post_shortcode"), item.get("num"))
|
|
|
|
|
|
def seed_archive_db(db: Path, existing: set[tuple[str, int]],
|
|
live: list[dict], kind: str) -> int:
|
|
"""
|
|
Mark everything already held as downloaded, so a fetch into an empty
|
|
directory pulls only what is missing.
|
|
|
|
`live` is the metadata of one listing pass — the pass we have to make
|
|
anyway — each entry carrying at least `media_id` plus the shortcode fields
|
|
`live_key` needs. Seeding costs no additional Instagram requests, and needs
|
|
only a *listing* of the archive, never its contents.
|
|
"""
|
|
import sqlite3
|
|
|
|
db.parent.mkdir(parents=True, exist_ok=True)
|
|
con = sqlite3.connect(db)
|
|
con.execute(ARCHIVE_SCHEMA)
|
|
rows = [
|
|
(ARCHIVE_KEY(item["media_id"]),)
|
|
for item in live
|
|
if live_key(item, kind) in existing
|
|
]
|
|
con.executemany("INSERT OR IGNORE INTO archive (entry) VALUES (?)", rows)
|
|
con.commit()
|
|
con.close()
|
|
return len(rows)
|
|
|
|
|
|
def probe_live(src: Source, config: Path, cookies: str) -> list[dict]:
|
|
"""
|
|
One metadata-only listing pass. `sleep` is forced to 0 because it otherwise
|
|
applies per *file* even with no download — 2275 files at 1-3s each is over
|
|
an hour for a single profile.
|
|
"""
|
|
out = subprocess.run(
|
|
["gallery-dl", "-j", "--config", str(config),
|
|
"--cookies-from-browser", cookies, "-o", "sleep=0", src.url],
|
|
capture_output=True, text=True, check=True,
|
|
)
|
|
items: list[dict] = []
|
|
|
|
def walk(node):
|
|
if isinstance(node, dict):
|
|
if "media_id" in node and "shortcode" in node:
|
|
items.append(node)
|
|
for value in node.values():
|
|
walk(value)
|
|
elif isinstance(node, list):
|
|
for value in node:
|
|
walk(value)
|
|
|
|
walk(json.loads(out.stdout))
|
|
return items
|
|
|
|
|
|
ALL_KINDS = ("posts", "reels", "stories", "highlights")
|
|
|
|
# Stories cannot be backfilled and expire in 24h, so a run that only wants
|
|
# stories is both cheap and the one worth scheduling daily.
|
|
STORIES_ONLY = {"stories"}
|
|
|
|
|
|
class SyncState:
|
|
"""
|
|
What has already been spent against `instagram.com`.
|
|
|
|
Exists because nothing else in this tool has any memory: every invocation
|
|
used to start from zero and happily re-enumerate profiles it had listed
|
|
minutes earlier. That is what suspended the account — the listing passes,
|
|
not the downloads.
|
|
|
|
Two facts are tracked per source:
|
|
|
|
seeded the skip-archive has been primed from the archive listing.
|
|
This is a ONE-TIME bootstrap: afterwards the archive DB records
|
|
every item gallery-dl has seen, so the source never needs
|
|
probing again. This is the single biggest request saving here.
|
|
fetched when it was last downloaded, so a re-run soon after is refused
|
|
rather than silently repeating the whole pass.
|
|
"""
|
|
|
|
VERSION = 1
|
|
|
|
def __init__(self, path: Path):
|
|
self.path = path
|
|
self.data = {"version": self.VERSION, "sources": {}}
|
|
if path.is_file():
|
|
try:
|
|
loaded = json.loads(path.read_text())
|
|
if loaded.get("version") == self.VERSION:
|
|
self.data = loaded
|
|
except Exception:
|
|
pass # a corrupt state file must never block a sync
|
|
|
|
def _entry(self, url: str) -> dict:
|
|
return self.data.setdefault("sources", {}).setdefault(url, {})
|
|
|
|
def needs_seed(self, url: str) -> bool:
|
|
return not self._entry(url).get("seeded")
|
|
|
|
def mark_seeded(self, url: str, stamp: str) -> None:
|
|
self._entry(url)["seeded"] = stamp
|
|
|
|
def last_fetch(self, url: str) -> str | None:
|
|
return self._entry(url).get("fetched")
|
|
|
|
def mark_fetched(self, url: str, stamp: str) -> None:
|
|
self._entry(url)["fetched"] = stamp
|
|
|
|
def save(self) -> None:
|
|
self.path.parent.mkdir(parents=True, exist_ok=True)
|
|
self.path.write_text(json.dumps(self.data, indent=1, sort_keys=True))
|
|
|
|
|
|
def hours_since(stamp: str | None, now: float) -> float:
|
|
"""Hours between an ISO stamp and `now`; infinite when never."""
|
|
if not stamp:
|
|
return float("inf")
|
|
try:
|
|
then = dt.datetime.fromisoformat(stamp)
|
|
except ValueError:
|
|
return float("inf")
|
|
if then.tzinfo is None:
|
|
then = then.replace(tzinfo=dt.timezone.utc)
|
|
return (now - then.timestamp()) / 3600.0
|
|
|
|
|
|
def plan_source(src: Source, state: SyncState, now: float,
|
|
min_interval: float) -> tuple[bool, bool, str]:
|
|
"""
|
|
Decide what a source needs: (fetch, seed, reason).
|
|
|
|
Seeding is skipped once done, and skipped entirely for stories — a story
|
|
cannot exist in the archive before it is fetched, so there is nothing to
|
|
seed from, and probing would double the request cost of the cheapest
|
|
surface we have.
|
|
"""
|
|
since = hours_since(state.last_fetch(src.url), now)
|
|
if since < min_interval:
|
|
return (False, False, f"fetched {since:.1f}h ago, under the "
|
|
f"{min_interval:g}h floor")
|
|
if src.kind == "stories":
|
|
return (True, False, "stories: no seed needed")
|
|
if state.needs_seed(src.url):
|
|
return (True, True, "first run: seeding from the archive listing")
|
|
return (True, False, "already seeded; the skip-archive knows what we hold")
|
|
|
|
|
|
class ProbeCache:
|
|
"""
|
|
Listing-pass results, kept so an interrupted run does not pay for them
|
|
twice. Yesterday an aborted sync re-enumerated five profiles on restart.
|
|
"""
|
|
|
|
def __init__(self, path: Path, ttl_hours: float):
|
|
self.path = path
|
|
self.ttl = ttl_hours
|
|
self.data: dict = {}
|
|
if path.is_file():
|
|
try:
|
|
self.data = json.loads(path.read_text())
|
|
except Exception:
|
|
self.data = {}
|
|
|
|
def get(self, url: str, now: float) -> list[dict] | None:
|
|
entry = self.data.get(url)
|
|
if not entry or hours_since(entry.get("at"), now) > self.ttl:
|
|
return None
|
|
return entry.get("items")
|
|
|
|
def put(self, url: str, items: list[dict], stamp: str) -> None:
|
|
# Only the fields seeding needs, so the cache stays small.
|
|
self.data[url] = {"at": stamp, "items": [
|
|
{k: i.get(k) for k in ("shortcode", "post_shortcode", "num", "media_id")}
|
|
for i in items
|
|
]}
|
|
|
|
def save(self) -> None:
|
|
self.path.parent.mkdir(parents=True, exist_ok=True)
|
|
self.path.write_text(json.dumps(self.data))
|
|
|
|
|
|
def rsync_command(staging: Path, dest: str, dry_run: bool) -> list[str]:
|
|
"""
|
|
Publish a staging tree into the archive.
|
|
|
|
`--ignore-existing` is not an optimisation, it is the safety property: the
|
|
archive deliberately outlives Instagram (posts exist here that Instagram no
|
|
longer serves), so publishing must only ever *add*. No `--delete`, and
|
|
nothing already present is overwritten — including sidecars, which get
|
|
rewritten on every run and would otherwise churn the synced share.
|
|
|
|
`dest` may be a local path or any rsync destination (`user@host:/path`),
|
|
because the archive usually is not writable from the fetch host.
|
|
"""
|
|
cmd = ["rsync", "-a", "--ignore-existing", "--partial", "--info=stats2",
|
|
# Belt and braces: the config lives outside staging, but nothing
|
|
# resembling tooling output should ever reach the archive. Archive
|
|
# sidecars are always "<date>_<user> - <code>.json", so none of
|
|
# these can match real content.
|
|
"--exclude", "gdl-sync*.json",
|
|
"--exclude", "*.gdl-config.json",
|
|
"--exclude", ".gdl-*",
|
|
"--exclude", "*.sqlite", "--exclude", "*.db"]
|
|
if dry_run:
|
|
cmd.append("--dry-run")
|
|
# Trailing slash: copy the *contents* of staging into dest.
|
|
cmd += [f"{staging}/", dest if dest.endswith("/") else dest + "/"]
|
|
return cmd
|
|
|
|
|
|
def publish(staging: Path, dest: str, dry_run: bool) -> int:
|
|
if not any(staging.iterdir()):
|
|
print(" nothing staged; skipping publish")
|
|
return 0
|
|
cmd = rsync_command(staging, dest, dry_run)
|
|
print(" " + " ".join(cmd))
|
|
return subprocess.run(cmd).returncode
|
|
|
|
|
|
def run_post_urls(args) -> int:
|
|
"""
|
|
Fetch one or more individual posts/reels by URL -- an ad hoc pull outside
|
|
the tracked profile list, e.g. a link shared from some other account. Each
|
|
is a single request, not a recurring surface, so there is no archive-db
|
|
seeding and no --min-interval floor to plan around.
|
|
"""
|
|
config = build_config(args.rate, list(args.sleep_request),
|
|
list(args.sleep), abort=0)
|
|
args.staging.mkdir(parents=True, exist_ok=True)
|
|
config_path = args.staging.parent / f"{args.staging.name}.gdl-config.json"
|
|
|
|
sources = [Source("post", url, "", "post") for url in args.post_url]
|
|
|
|
print(f"post-url : {len(sources)} to fetch")
|
|
print(f"pacing : {args.sleep_request[0]}-{args.sleep_request[1]}s between "
|
|
f"requests, rate cap {args.rate}")
|
|
print(f"staging : {args.staging}")
|
|
print(f"publish : {args.publish}")
|
|
print()
|
|
|
|
if not args.execute:
|
|
for src in sources:
|
|
print(f" {src.url}")
|
|
print()
|
|
print(" " + " ".join(rsync_command(args.staging, args.publish, True)))
|
|
print("\ndry run; nothing fetched. pass --execute to run.")
|
|
return 0
|
|
|
|
config_path.write_text(json.dumps(config, indent=2))
|
|
failures = 0
|
|
for src in sources:
|
|
print(f"==> {src.url}")
|
|
cmd = gdl_command(src, args.staging, config_path, args.cookies,
|
|
args.archive_db)
|
|
result = subprocess.run(cmd)
|
|
if result.returncode != 0:
|
|
failures += 1
|
|
print(f" FAILED (exit {result.returncode})", file=sys.stderr)
|
|
|
|
print("\n==> publish")
|
|
if publish(args.staging, args.publish, dry_run=False) != 0:
|
|
failures += 1
|
|
|
|
print(f"\ndone; {failures} step(s) failed")
|
|
return 1 if failures else 0
|
|
|
|
|
|
def main() -> int:
|
|
# A sync runs for hours and is normally watched through a redirected log,
|
|
# where Python's block buffering would withhold progress until it happened
|
|
# to flush -- and the gallery-dl subprocesses write to the same descriptor
|
|
# unbuffered, so the log would also interleave out of order.
|
|
sys.stdout.reconfigure(line_buffering=True)
|
|
sys.stderr.reconfigure(line_buffering=True)
|
|
|
|
ap = argparse.ArgumentParser(description=__doc__, epilog=EXAMPLES,
|
|
formatter_class=argparse.RawDescriptionHelpFormatter)
|
|
ap.add_argument("--index",
|
|
help="existing archive listing: a local root, or the "
|
|
"viewer's base URL (only a FILE LISTING is needed, "
|
|
"never the contents). Required unless --post-url")
|
|
ap.add_argument("--publish", required=True,
|
|
help="rsync destination for fetched files; a local path or "
|
|
"user@host:/path")
|
|
ap.add_argument("--staging", type=Path, required=True,
|
|
help="local scratch directory gallery-dl writes into")
|
|
g = ap.add_mutually_exclusive_group(required=True)
|
|
g.add_argument("--profile", action="append", default=[],
|
|
help="profile to sync; repeatable")
|
|
g.add_argument("--all", action="store_true", help="every profile on disk")
|
|
g.add_argument("--urls-file", type=Path,
|
|
help="file of Instagram profile URLs or usernames, one per "
|
|
"line; # comments and blank lines allowed")
|
|
g.add_argument("--post-url", action="append", default=[],
|
|
help="fetch one post or reel by URL (e.g. "
|
|
"https://www.instagram.com/p/SHORTCODE/), filed under "
|
|
"its owner's account like any other post; repeatable. "
|
|
"A one-off fetch outside the tracked profile list: no "
|
|
"archive-db seeding, no --min-interval floor")
|
|
ap.add_argument("--cookies", default="chrome:/home/matt/.config/google-chrome-devtools",
|
|
help="gallery-dl --cookies-from-browser value")
|
|
ap.add_argument("--archive-db", type=Path, default=None,
|
|
help="gallery-dl skip-archive sqlite path")
|
|
ap.add_argument("--rate", default="1M", help="per-download rate cap")
|
|
ap.add_argument("--sleep-request", nargs=2, type=float, default=[6.0, 10.0],
|
|
metavar=("MIN", "MAX"))
|
|
ap.add_argument("--sleep", nargs=2, type=float, default=[3.0, 6.0],
|
|
metavar=("MIN", "MAX"))
|
|
ap.add_argument("--only", default=",".join(ALL_KINDS),
|
|
help="comma-separated surfaces to sync: "
|
|
"posts,reels,stories,highlights. Use --only stories "
|
|
"for the cheap daily run.")
|
|
ap.add_argument("--min-interval", type=float, default=20.0, metavar="HOURS",
|
|
help="refuse to re-fetch a source touched more recently "
|
|
"than this (default 20h); the guard that makes a "
|
|
"restart cheap instead of a repeat")
|
|
ap.add_argument("--max-sources", type=int, default=0, metavar="N",
|
|
help="hard ceiling on sources touched in one run "
|
|
"(0 = no limit)")
|
|
ap.add_argument("--abort", type=int, default=0, metavar="N",
|
|
help="stop enumerating posts/reels after N consecutive "
|
|
"already-archived FILES (0 = walk everything, the "
|
|
"default). 50 is a safe routine value; it cuts the "
|
|
"per-run listing cost by roughly 85%%, at the price "
|
|
"of no longer noticing edited carousels")
|
|
ap.add_argument("--probe-ttl", type=float, default=24.0, metavar="HOURS",
|
|
help="reuse cached listing results younger than this")
|
|
ap.add_argument("--force", action="store_true",
|
|
help="ignore --min-interval and the probe cache")
|
|
mode = ap.add_mutually_exclusive_group()
|
|
mode.add_argument("--dry-run", action="store_true", default=True,
|
|
help="print the plan and the config; default")
|
|
mode.add_argument("--execute", action="store_true",
|
|
help="actually run gallery-dl")
|
|
args = ap.parse_args()
|
|
|
|
if not shutil.which("gallery-dl"):
|
|
print("gallery-dl not on PATH", file=sys.stderr)
|
|
return 2
|
|
if not shutil.which("rsync"):
|
|
print("rsync not on PATH", file=sys.stderr)
|
|
return 2
|
|
|
|
if args.post_url:
|
|
return run_post_urls(args)
|
|
|
|
if not args.index:
|
|
print("--index is required unless --post-url is given", file=sys.stderr)
|
|
return 2
|
|
|
|
index = ArchiveIndex(args.index)
|
|
names = index.profiles()
|
|
if args.urls_file:
|
|
if not args.urls_file.is_file():
|
|
print(f"urls file not found: {args.urls_file}", file=sys.stderr)
|
|
return 2
|
|
wanted = read_urls_file(args.urls_file)
|
|
if not wanted:
|
|
print(f"no usable profiles in {args.urls_file}", file=sys.stderr)
|
|
return 2
|
|
print(f"read {len(wanted)} profile(s) from {args.urls_file}")
|
|
selected = [Profile(p) for p in wanted]
|
|
elif args.profile:
|
|
for p in args.profile:
|
|
if p not in names:
|
|
print(f"note: {p} is not in the index yet; it will be created")
|
|
selected = [Profile(p) for p in args.profile]
|
|
else:
|
|
selected = [Profile(p) for p in sorted(names)]
|
|
|
|
config = build_config(args.rate, list(args.sleep_request),
|
|
list(args.sleep), args.abort)
|
|
args.staging.mkdir(parents=True, exist_ok=True)
|
|
# Deliberately a SIBLING of the staging directory, not inside it: staging is
|
|
# rsynced wholesale into the archive, and a dry run caught this file being
|
|
# published to the archive root.
|
|
config_path = args.staging.parent / f"{args.staging.name}.gdl-config.json"
|
|
|
|
kinds = {k.strip() for k in args.only.split(",") if k.strip()}
|
|
unknown = kinds - set(ALL_KINDS)
|
|
if unknown:
|
|
print(f"unknown surface(s): {', '.join(sorted(unknown))}", file=sys.stderr)
|
|
return 2
|
|
|
|
state_path = (args.archive_db.with_suffix(".state.json") if args.archive_db
|
|
else args.staging.parent / f"{args.staging.name}.state.json")
|
|
state = SyncState(state_path)
|
|
now = dt.datetime.now(dt.timezone.utc)
|
|
now_ts, stamp = now.timestamp(), now.isoformat()
|
|
min_interval = 0.0 if args.force else args.min_interval
|
|
|
|
plan: list[tuple[Profile, Source, bool]] = []
|
|
skipped = 0
|
|
for prof in selected:
|
|
for src in prof.sources(kinds):
|
|
fetch, seed, reason = plan_source(src, state, now_ts, min_interval)
|
|
if not fetch:
|
|
skipped += 1
|
|
print(f" skip {prof.user}/{src.kind}: {reason}")
|
|
continue
|
|
if args.max_sources and len(plan) >= args.max_sources:
|
|
skipped += 1
|
|
continue
|
|
plan.append((prof, src, seed))
|
|
|
|
print(f"profiles : {len(selected)}")
|
|
print(f"surfaces : {','.join(k for k in ALL_KINDS if k in kinds)}")
|
|
print(f"sources : {len(plan)} to sync, {skipped} skipped")
|
|
print(f"pacing : {args.sleep_request[0]}-{args.sleep_request[1]}s between "
|
|
f"requests, rate cap {args.rate}")
|
|
print(f"staging : {args.staging}")
|
|
print(f"publish : {args.publish}")
|
|
print()
|
|
|
|
if not args.execute:
|
|
for prof, src, seed in plan:
|
|
dest = src.directory or "(per-highlight)"
|
|
note = " [will seed]" if seed else ""
|
|
print(f" {prof.user:<20} {src.kind:<11} -> {dest}{note}")
|
|
print()
|
|
print(" " + " ".join(rsync_command(args.staging, args.publish, True)))
|
|
print("\ndry run; nothing fetched. pass --execute to run.")
|
|
return 0
|
|
|
|
config_path.write_text(json.dumps(config, indent=2))
|
|
probes = ProbeCache(state_path.with_suffix(".probes.json"),
|
|
0.0 if args.force else args.probe_ttl)
|
|
failures = 0
|
|
|
|
for prof, src, seed in plan:
|
|
print(f"==> {prof.user} / {src.kind}")
|
|
stage_dir = args.staging / (src.directory or ".")
|
|
stage_dir.mkdir(parents=True, exist_ok=True)
|
|
|
|
# Prime the skip-archive from what the archive already holds, so
|
|
# fetching into an empty staging directory pulls only what is missing.
|
|
# Done once per source, ever: afterwards the archive DB records
|
|
# everything gallery-dl has seen and no listing pass is needed.
|
|
if seed and args.archive_db:
|
|
try:
|
|
live = probes.get(src.url, now_ts)
|
|
if live is None:
|
|
live = probe_live(src, config_path, args.cookies)
|
|
probes.put(src.url, live, stamp)
|
|
probes.save()
|
|
else:
|
|
print(f" reusing {len(live)} cached listing items")
|
|
held = index_existing(index.listing(prof.user))
|
|
seeded = seed_archive_db(args.archive_db, held, live,
|
|
src.subcategory)
|
|
print(f" seeded {seeded} of {len(live)} live items")
|
|
state.mark_seeded(src.url, stamp)
|
|
state.save()
|
|
except subprocess.CalledProcessError as exc:
|
|
failures += 1
|
|
print(f" probe FAILED: {exc}", file=sys.stderr)
|
|
continue
|
|
|
|
cmd = gdl_command(src, args.staging, config_path, args.cookies,
|
|
args.archive_db)
|
|
result = subprocess.run(cmd)
|
|
if result.returncode != 0:
|
|
failures += 1
|
|
# Keep going: one private or renamed profile must not abort the run.
|
|
print(f" FAILED (exit {result.returncode})", file=sys.stderr)
|
|
else:
|
|
# Recorded even for an empty fetch: the request was still spent.
|
|
state.mark_fetched(src.url, stamp)
|
|
state.save()
|
|
|
|
# Publish once, at the end, so a partially-fetched profile never reaches
|
|
# the archive mid-run. Only ever adds -- see rsync_command.
|
|
print("\n==> publish")
|
|
if publish(args.staging, args.publish, dry_run=False) != 0:
|
|
failures += 1
|
|
|
|
print(f"\ndone; {failures} step(s) failed")
|
|
return 1 if failures else 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main())
|