Files
instaarchive-viewer/scripts/gdl-sync.py
T
ergosteurandClaude Sonnet 5 0753395391 feat: add --post-url to fetch an arbitrary single post or reel
Lets an out-of-band link (shared by someone, not one of the tracked
profiles) be pulled in directly by URL, filed under its owner's account
like any other post. Bypasses profile planning, archive-db seeding, and
the --min-interval floor entirely, since it's a single request rather
than a recurring surface to budget against.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_011qAds5qr7nZRq5R4yAuxUk
2026-08-26 20:43:38 -04:00

913 lines
37 KiB
Python
Executable File

#!/usr/bin/env python3
"""
Fetch Instagram profiles into the archive layout using gallery-dl.
The CLI replacement for the JDownloader2 workflow. See docs/gallery-dl.md for
the measurements behind every choice here — especially the safety model, which
is the reason this script exists in this shape rather than a simpler one.
The fetch host needs no copy of the archive. It stages locally and rsyncs
afterwards; what it already holds is learned from a *file listing* alone
(`--index`), which the viewer's own API serves.
Usage:
./scripts/gdl-sync.py --index https://instaarchive.ergosteur.com \\
--staging /var/tmp/gdl --publish user@host:/path/to/archives \\
--urls-file artms_account_links.txt --dry-run
# ...then swap --dry-run for --execute. --index also accepts a local path,
# and --profile / --all work instead of --urls-file.
Always --dry-run first: it prints the plan, and the publish step it reports is
the one that would touch the archive.
Run it from the host whose public IP matches the browser the cookie came from;
using the cookie from elsewhere is what session-hijack detection looks for.
"""
from __future__ import annotations
import argparse
import datetime as dt
import json
import os
import re
import shutil
import subprocess
import sys
from dataclasses import dataclass, field
from pathlib import Path
# --------------------------------------------------------------------------
# Archive layout
# --------------------------------------------------------------------------
# Mirrors src/lib/archive-grouping.ts. Instagram usernames cannot contain
# spaces, which is what makes the username separable from a highlight title.
RE_HIGHLIGHT = re.compile(r"^story highlights - ([^ ]+) - (.+)$")
RE_STORIES = re.compile(r"^story - ([^ ]+)$")
RE_REELS = re.compile(r"^([^ ]+) - reels$")
DATE_FMT = "{date:Olocal/%Y-%m-%d}"
"""Local-time date. JD2 stamped US Eastern, NOT UTC (0/212 mismatches vs 19 for
UTC). `Olocal` is DST-aware per timestamp. The trailing separator must be
omitted or it lands in the strftime format and sanitises to an underscore."""
POST_STEM = DATE_FMT + "_{username} - {post_shortcode}"
ITEM_STEM = DATE_FMT + "_{username} - {shortcode}"
@dataclass
class Source:
"""One gallery-dl invocation: a URL fetched into a specific directory."""
kind: str # posts | reels | stories | highlights
url: str
directory: str # relative to the archives root
subcategory: str # gallery-dl config key
title: str | None = None # highlight title, when known
@dataclass
class Profile:
user: str
existing: dict[str, str] = field(default_factory=dict) # kind -> dirname
def sources(self, kinds: set[str]) -> list[Source]:
u = self.user
base = f"https://www.instagram.com/{u}"
all_sources = [
Source("posts", f"{base}/posts/", u, "posts"),
Source("reels", f"{base}/reels/", f"{u} - reels", "reels"),
# Stories expire after 24h, so these can only ever be captured
# live. There is no backfill and no re-fetch -- which is why they
# are the one surface worth visiting daily.
Source("stories", f"https://www.instagram.com/stories/{u}/",
f"story - {u}", "stories"),
# Highlight directories embed the title, which gallery-dl only
# learns mid-extraction -- so this one source fans out into many
# directories and is handled with a directory format string.
Source("highlights", f"{base}/highlights", "", "highlights"),
]
return [s for s in all_sources if s.kind in kinds]
def scan_archives(root: Path) -> dict[str, Profile]:
"""Group existing directories into profiles, as the server does."""
profiles: dict[str, Profile] = {}
def get(user: str) -> Profile:
return profiles.setdefault(user, Profile(user))
for entry in sorted(os.listdir(root)):
if not (root / entry).is_dir() or entry.startswith("."):
continue
if m := RE_HIGHLIGHT.match(entry):
get(m.group(1)).existing.setdefault("highlights", entry)
elif m := RE_STORIES.match(entry):
get(m.group(1)).existing["stories"] = entry
elif m := RE_REELS.match(entry):
get(m.group(1)).existing["reels"] = entry
else:
get(entry).existing["posts"] = entry
return profiles
RE_PROFILE_URL = re.compile(
r"^(?:https?://)?(?:www\.)?instagram\.com/(?P<user>[^/?#\s]+)/?", re.I)
# Path segments that are Instagram features, not profiles. A line like
# ".../p/ABC123/" names a post, and treating "p" as a username would silently
# sync nothing under a nonsense directory.
RESERVED_SEGMENTS = {
"p", "reel", "reels", "stories", "explore", "accounts", "direct",
"tv", "s", "invites", "challenge", "about", "developer",
}
def read_urls_file(path: Path) -> list[str]:
"""
Read profile URLs (or bare usernames) from a file, one per line.
Written for hand-maintained lists: blank lines are skipped, `#` starts a
comment, and either a full URL or a bare username works. Order is kept and
duplicates dropped, so a list can be appended to without care.
"""
users: list[str] = []
seen: set[str] = set()
for lineno, raw in enumerate(path.read_text().splitlines(), 1):
line = raw.split("#", 1)[0].strip()
if not line:
continue
m = RE_PROFILE_URL.match(line)
user = m.group("user") if m else line.strip("/")
if not user or "/" in user or " " in user:
print(f"{path}:{lineno}: cannot read a username from {raw.strip()!r}",
file=sys.stderr)
continue
if user.lower() in RESERVED_SEGMENTS:
print(f"{path}:{lineno}: {user!r} is an Instagram path, not a "
f"profile — skipping", file=sys.stderr)
continue
if user in seen:
continue
seen.add(user)
users.append(user)
return users
class ArchiveIndex:
"""
What the archive already holds, as filenames only.
Deliberately never reads file *contents*, so the fetch host does not need a
copy of the archive — it can stage locally and rsync afterwards. Backed
either by a local directory or by the viewer's own API, which already
serves exactly this listing and is the cheaper option when the archive
lives on network storage (a full walk there took ~52s).
"""
def __init__(self, source: str):
self.remote = source.startswith(("http://", "https://"))
self.source = source.rstrip("/") if self.remote else None
self.root = None if self.remote else Path(source)
if self.root and not self.root.is_dir():
raise SystemExit(f"archive index not found: {source}")
self._cache: dict[str, list[str]] = {}
def _get(self, path: str):
from urllib.request import urlopen
with urlopen(f"{self.source}{path}", timeout=60) as resp:
return json.load(resp)
def profiles(self) -> set[str]:
if self.remote:
return {a["name"] for a in self._get("/api/archives")}
return set(scan_archives(self.root))
def listing(self, user: str) -> list[str]:
"""Every filename belonging to a profile, across all its sidecars."""
if user in self._cache:
return self._cache[user]
names: list[str] = []
if self.remote:
try:
data = self._get(f"/api/archives/{user}/files")
except Exception:
data = []
files = data if isinstance(data, list) else data.get("files", [])
names = [f["path"] for f in files]
else:
prof = scan_archives(self.root).get(user)
for dirname in (prof.existing.values() if prof else ()):
d = self.root / dirname
if d.is_dir():
names += [f"{dirname}/{n}" for n in os.listdir(d)]
self._cache[user] = names
return names
# --------------------------------------------------------------------------
# gallery-dl configuration
# --------------------------------------------------------------------------
def build_config(rate: str, sleep_request: list[float],
sleep: list[float], abort: int = 0) -> dict:
"""
The config is generated rather than checked in so the safety-critical
options cannot drift out of sync with the docs.
`api: rest` is the single most important line in this file. The graphql
backend issues one request PER POST for every video and carousel, which is
the pattern that got this account banned once already.
"""
caption_pp = {
"name": "metadata",
"event": "post",
"mode": "custom",
"content-format": "{description}",
"extension": "txt",
# JD2 wrote no .txt when the caption was empty; "empty": false (the
# default) reproduces that.
}
meta_pp = {
"name": "metadata",
"event": "post",
"mode": "json",
# `include`, NOT `fields` -- `fields` applies to mode:custom and
# silently does nothing here, dumping audio_user blobs that contain
# unrelated users' profile picture URLs.
"include": [
"post_shortcode", "post_id", "type", "date", "post_date",
"username", "fullname", "owner_id", "description", "count",
"likes", "post_url", "sidecar_shortcode",
],
}
def post_like(stem: str) -> dict:
"""Naming for surfaces whose unit is a post (posts, reels)."""
skip: dict = {}
if abort:
# Stop enumerating once `abort` consecutive files are already in
# the skip-archive. The listing pass -- not the downloading -- is
# what costs `instagram.com` requests, and it otherwise walks the
# whole profile every run to find three new posts.
#
# Safe here only because the REST listing is strictly
# reverse-chronological: the web grid hoists pinned posts to the
# front, but this endpoint does not (measured 2026-08-20), so old
# posts never appear before new ones.
#
# Counted in FILES, not posts, so it must clear the largest
# already-held carousel -- 22 media for one real post in this
# archive. It also means edited carousels (test case 15) stop
# being noticed, so a full sweep is still worth running
# occasionally.
skip["skip"] = f"abort:{abort}"
return {
**skip,
# `sidecar_shortcode` is set only for carousels, so it is the
# carousel discriminator. First matching condition wins.
"filename": {
"sidecar_shortcode and count >= 10":
stem + " - {num:02}.{extension}",
"sidecar_shortcode":
stem + " - {num}.{extension}",
"":
stem + ".{extension}",
},
"postprocessors": [
{**caption_pp, "filename": stem + ".txt"},
{**meta_pp, "filename": stem + ".json"},
],
}
def item_like(stem: str) -> dict:
"""
Naming for surfaces whose unit is an item inside a reel (stories,
highlights). `{shortcode}` is per item; `{post_shortcode}` is the
reel's id and is shared by every item in it.
The media filename uses the per-item shortcode, but the sidecar cannot:
it runs at `event: post`, where the kwdict describes the *reel* and has
no `shortcode` at all -- which silently formatted as the literal
"None", producing one "<date>_<user> - None.json" per reel. It is keyed
by `post_shortcode` instead, and is genuinely reel-level data (the
reel's own date and item count); per-item dates live in the media
filenames, which is the more precise source anyway.
"""
return {
"filename": stem + ".{extension}",
"postprocessors": [
{**meta_pp,
"filename": DATE_FMT + "_{username} - {post_shortcode}.json"},
],
}
return {
"extractor": {
"base-directory": ".",
"instagram": {
"api": "rest", # never "graphql" -- see docstring
"sleep-request": sleep_request,
"sleep": sleep,
# The CDN does rate-limit: a first run at 3M/1-3s drew
# '429 Too Many Requests' from scontent-*.cdninstagram.com and
# lost two videos. Back off hard rather than retry fast.
"sleep-429": 120.0,
"retries": 8,
"videos": True,
"include": "", # never "all"; sources are explicit
# Directory is forced per-invocation with -D, because a reels
# tab returns collab reels owned by OTHER accounts and
# {username} would scatter them into the wrong profile.
"directory": [],
"posts": post_like(POST_STEM),
"reels": post_like(POST_STEM),
# Ad hoc single-post fetches (--post-url) go through gallery-dl's
# own post/reel extractor instead of a profile listing, so the
# owning account is never known ahead of time -- only mid-
# extraction, same reasoning as highlights below.
"post": {**post_like(POST_STEM), "directory": ["{username}"]},
"reel": {**post_like(POST_STEM), "directory": ["{username}"]},
"stories": item_like(ITEM_STEM),
"highlights": {
**item_like(ITEM_STEM),
# The only surface that must derive its own directory,
# since the title is not known until extraction.
"directory": ["story highlights - {username} - {highlight_title}"],
},
},
},
# `retries` here is the CDN-side counterpart to sleep-429 above.
"downloader": {"http": {"rate": rate, "retries": 8}},
"output": {"mode": "null"},
}
# --------------------------------------------------------------------------
# Planning and execution
# --------------------------------------------------------------------------
def gdl_command(src: Source, staging: Path, config: Path, cookies: str,
archive_db: Path | None) -> list[str]:
cmd = [
"gallery-dl",
"--config", str(config),
"--cookies-from-browser", cookies,
]
if archive_db:
# Without a seeded skip-archive, staging is empty and every file is
# re-downloaded; see seed_archive_db.
cmd += ["--download-archive", str(archive_db)]
# Forced destination -- never `{username}` -- because a reels tab returns
# collab reels owned by other accounts, which would otherwise be filed
# under the wrong profile. Highlights and ad hoc single-post fetches are
# the exception: their directory is only known mid-extraction (a title, or
# the post's owner), so the config formats it instead.
dest = (staging if src.subcategory in ("highlights", "post", "reel")
else staging / src.directory)
cmd += ["--destination", str(dest)]
cmd.append(src.url)
return cmd
# gallery-dl keys its skip-archive on `archive_prefix + archive_fmt`, which for
# this extractor is the literal "instagram" followed by the per-media numeric
# pk (`instagram.py:25`, `job.py:713-719`). Verified against a real run: a
# 3-image carousel produced 3 rows, one per item.
ARCHIVE_KEY = "instagram{}".format
ARCHIVE_SCHEMA = "CREATE TABLE IF NOT EXISTS archive (entry TEXT PRIMARY KEY)"
RE_ARCHIVED = re.compile(
r"^(\d{4}-\d{2}-\d{2})_(.+?) - ([A-Za-z0-9_-]+?)(?: - (\d+))?\.(\w+)$")
NON_MEDIA = {"txt", "json"}
def index_existing(listing: list[str]) -> set[tuple[str, int]]:
"""
Reduce a flat list of filenames to the (shortcode, index) pairs already
held. Only names matter — never the bytes — which is what lets the sync run
on a host that has no copy of the archive.
"""
have: set[tuple[str, int]] = set()
for name in listing:
m = RE_ARCHIVED.match(name.rsplit("/", 1)[-1])
if not m or m.group(5).lower() in NON_MEDIA:
continue
# An absent index means a single-media post, which is index 1 — the
# same normalisation the viewer's EXPORT_RE applies.
have.add((m.group(3), int(m.group(4) or 1)))
return have
def live_key(item: dict, kind: str) -> tuple[str, int]:
"""
The (shortcode, index) a live item *would* be filed under, mirroring the
filename template exactly.
The two surfaces disagree about which shortcode identifies a file, and
getting this wrong silently seeds almost nothing:
posts/reels filed under {post_shortcode} — for a carousel, each
child item ALSO has its own `shortcode`, which is not
what appears in the filename.
stories/highlights filed under the per-item {shortcode}, because
`post_shortcode` there is the containing reel's id and
is shared by every item in it.
"""
if kind in ("stories", "highlights"):
return (item.get("shortcode"), 1)
return (item.get("post_shortcode"), item.get("num"))
def seed_archive_db(db: Path, existing: set[tuple[str, int]],
live: list[dict], kind: str) -> int:
"""
Mark everything already held as downloaded, so a fetch into an empty
directory pulls only what is missing.
`live` is the metadata of one listing pass — the pass we have to make
anyway — each entry carrying at least `media_id` plus the shortcode fields
`live_key` needs. Seeding costs no additional Instagram requests, and needs
only a *listing* of the archive, never its contents.
"""
import sqlite3
db.parent.mkdir(parents=True, exist_ok=True)
con = sqlite3.connect(db)
con.execute(ARCHIVE_SCHEMA)
rows = [
(ARCHIVE_KEY(item["media_id"]),)
for item in live
if live_key(item, kind) in existing
]
con.executemany("INSERT OR IGNORE INTO archive (entry) VALUES (?)", rows)
con.commit()
con.close()
return len(rows)
def probe_live(src: Source, config: Path, cookies: str) -> list[dict]:
"""
One metadata-only listing pass. `sleep` is forced to 0 because it otherwise
applies per *file* even with no download — 2275 files at 1-3s each is over
an hour for a single profile.
"""
out = subprocess.run(
["gallery-dl", "-j", "--config", str(config),
"--cookies-from-browser", cookies, "-o", "sleep=0", src.url],
capture_output=True, text=True, check=True,
)
items: list[dict] = []
def walk(node):
if isinstance(node, dict):
if "media_id" in node and "shortcode" in node:
items.append(node)
for value in node.values():
walk(value)
elif isinstance(node, list):
for value in node:
walk(value)
walk(json.loads(out.stdout))
return items
ALL_KINDS = ("posts", "reels", "stories", "highlights")
# Stories cannot be backfilled and expire in 24h, so a run that only wants
# stories is both cheap and the one worth scheduling daily.
STORIES_ONLY = {"stories"}
class SyncState:
"""
What has already been spent against `instagram.com`.
Exists because nothing else in this tool has any memory: every invocation
used to start from zero and happily re-enumerate profiles it had listed
minutes earlier. That is what suspended the account — the listing passes,
not the downloads.
Two facts are tracked per source:
seeded the skip-archive has been primed from the archive listing.
This is a ONE-TIME bootstrap: afterwards the archive DB records
every item gallery-dl has seen, so the source never needs
probing again. This is the single biggest request saving here.
fetched when it was last downloaded, so a re-run soon after is refused
rather than silently repeating the whole pass.
"""
VERSION = 1
def __init__(self, path: Path):
self.path = path
self.data = {"version": self.VERSION, "sources": {}}
if path.is_file():
try:
loaded = json.loads(path.read_text())
if loaded.get("version") == self.VERSION:
self.data = loaded
except Exception:
pass # a corrupt state file must never block a sync
def _entry(self, url: str) -> dict:
return self.data.setdefault("sources", {}).setdefault(url, {})
def needs_seed(self, url: str) -> bool:
return not self._entry(url).get("seeded")
def mark_seeded(self, url: str, stamp: str) -> None:
self._entry(url)["seeded"] = stamp
def last_fetch(self, url: str) -> str | None:
return self._entry(url).get("fetched")
def mark_fetched(self, url: str, stamp: str) -> None:
self._entry(url)["fetched"] = stamp
def save(self) -> None:
self.path.parent.mkdir(parents=True, exist_ok=True)
self.path.write_text(json.dumps(self.data, indent=1, sort_keys=True))
def hours_since(stamp: str | None, now: float) -> float:
"""Hours between an ISO stamp and `now`; infinite when never."""
if not stamp:
return float("inf")
try:
then = dt.datetime.fromisoformat(stamp)
except ValueError:
return float("inf")
if then.tzinfo is None:
then = then.replace(tzinfo=dt.timezone.utc)
return (now - then.timestamp()) / 3600.0
def plan_source(src: Source, state: SyncState, now: float,
min_interval: float) -> tuple[bool, bool, str]:
"""
Decide what a source needs: (fetch, seed, reason).
Seeding is skipped once done, and skipped entirely for stories — a story
cannot exist in the archive before it is fetched, so there is nothing to
seed from, and probing would double the request cost of the cheapest
surface we have.
"""
since = hours_since(state.last_fetch(src.url), now)
if since < min_interval:
return (False, False, f"fetched {since:.1f}h ago, under the "
f"{min_interval:g}h floor")
if src.kind == "stories":
return (True, False, "stories: no seed needed")
if state.needs_seed(src.url):
return (True, True, "first run: seeding from the archive listing")
return (True, False, "already seeded; the skip-archive knows what we hold")
class ProbeCache:
"""
Listing-pass results, kept so an interrupted run does not pay for them
twice. Yesterday an aborted sync re-enumerated five profiles on restart.
"""
def __init__(self, path: Path, ttl_hours: float):
self.path = path
self.ttl = ttl_hours
self.data: dict = {}
if path.is_file():
try:
self.data = json.loads(path.read_text())
except Exception:
self.data = {}
def get(self, url: str, now: float) -> list[dict] | None:
entry = self.data.get(url)
if not entry or hours_since(entry.get("at"), now) > self.ttl:
return None
return entry.get("items")
def put(self, url: str, items: list[dict], stamp: str) -> None:
# Only the fields seeding needs, so the cache stays small.
self.data[url] = {"at": stamp, "items": [
{k: i.get(k) for k in ("shortcode", "post_shortcode", "num", "media_id")}
for i in items
]}
def save(self) -> None:
self.path.parent.mkdir(parents=True, exist_ok=True)
self.path.write_text(json.dumps(self.data))
def rsync_command(staging: Path, dest: str, dry_run: bool) -> list[str]:
"""
Publish a staging tree into the archive.
`--ignore-existing` is not an optimisation, it is the safety property: the
archive deliberately outlives Instagram (posts exist here that Instagram no
longer serves), so publishing must only ever *add*. No `--delete`, and
nothing already present is overwritten — including sidecars, which get
rewritten on every run and would otherwise churn the synced share.
`dest` may be a local path or any rsync destination (`user@host:/path`),
because the archive usually is not writable from the fetch host.
"""
cmd = ["rsync", "-a", "--ignore-existing", "--partial", "--info=stats2",
# Belt and braces: the config lives outside staging, but nothing
# resembling tooling output should ever reach the archive. Archive
# sidecars are always "<date>_<user> - <code>.json", so none of
# these can match real content.
"--exclude", "gdl-sync*.json",
"--exclude", "*.gdl-config.json",
"--exclude", ".gdl-*",
"--exclude", "*.sqlite", "--exclude", "*.db"]
if dry_run:
cmd.append("--dry-run")
# Trailing slash: copy the *contents* of staging into dest.
cmd += [f"{staging}/", dest if dest.endswith("/") else dest + "/"]
return cmd
def publish(staging: Path, dest: str, dry_run: bool) -> int:
if not any(staging.iterdir()):
print(" nothing staged; skipping publish")
return 0
cmd = rsync_command(staging, dest, dry_run)
print(" " + " ".join(cmd))
return subprocess.run(cmd).returncode
def run_post_urls(args) -> int:
"""
Fetch one or more individual posts/reels by URL -- an ad hoc pull outside
the tracked profile list, e.g. a link shared from some other account. Each
is a single request, not a recurring surface, so there is no archive-db
seeding and no --min-interval floor to plan around.
"""
config = build_config(args.rate, list(args.sleep_request),
list(args.sleep), abort=0)
args.staging.mkdir(parents=True, exist_ok=True)
config_path = args.staging.parent / f"{args.staging.name}.gdl-config.json"
sources = [Source("post", url, "", "post") for url in args.post_url]
print(f"post-url : {len(sources)} to fetch")
print(f"pacing : {args.sleep_request[0]}-{args.sleep_request[1]}s between "
f"requests, rate cap {args.rate}")
print(f"staging : {args.staging}")
print(f"publish : {args.publish}")
print()
if not args.execute:
for src in sources:
print(f" {src.url}")
print()
print(" " + " ".join(rsync_command(args.staging, args.publish, True)))
print("\ndry run; nothing fetched. pass --execute to run.")
return 0
config_path.write_text(json.dumps(config, indent=2))
failures = 0
for src in sources:
print(f"==> {src.url}")
cmd = gdl_command(src, args.staging, config_path, args.cookies,
args.archive_db)
result = subprocess.run(cmd)
if result.returncode != 0:
failures += 1
print(f" FAILED (exit {result.returncode})", file=sys.stderr)
print("\n==> publish")
if publish(args.staging, args.publish, dry_run=False) != 0:
failures += 1
print(f"\ndone; {failures} step(s) failed")
return 1 if failures else 0
def main() -> int:
# A sync runs for hours and is normally watched through a redirected log,
# where Python's block buffering would withhold progress until it happened
# to flush -- and the gallery-dl subprocesses write to the same descriptor
# unbuffered, so the log would also interleave out of order.
sys.stdout.reconfigure(line_buffering=True)
sys.stderr.reconfigure(line_buffering=True)
ap = argparse.ArgumentParser(description=__doc__,
formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument("--index",
help="existing archive listing: a local root, or the "
"viewer's base URL (only a FILE LISTING is needed, "
"never the contents). Required unless --post-url")
ap.add_argument("--publish", required=True,
help="rsync destination for fetched files; a local path or "
"user@host:/path")
ap.add_argument("--staging", type=Path, required=True,
help="local scratch directory gallery-dl writes into")
g = ap.add_mutually_exclusive_group(required=True)
g.add_argument("--profile", action="append", default=[],
help="profile to sync; repeatable")
g.add_argument("--all", action="store_true", help="every profile on disk")
g.add_argument("--urls-file", type=Path,
help="file of Instagram profile URLs or usernames, one per "
"line; # comments and blank lines allowed")
g.add_argument("--post-url", action="append", default=[],
help="fetch one post or reel by URL (e.g. "
"https://www.instagram.com/p/SHORTCODE/), filed under "
"its owner's account like any other post; repeatable. "
"A one-off fetch outside the tracked profile list: no "
"archive-db seeding, no --min-interval floor")
ap.add_argument("--cookies", default="chrome:/home/matt/.config/google-chrome-devtools",
help="gallery-dl --cookies-from-browser value")
ap.add_argument("--archive-db", type=Path, default=None,
help="gallery-dl skip-archive sqlite path")
ap.add_argument("--rate", default="1M", help="per-download rate cap")
ap.add_argument("--sleep-request", nargs=2, type=float, default=[6.0, 10.0],
metavar=("MIN", "MAX"))
ap.add_argument("--sleep", nargs=2, type=float, default=[3.0, 6.0],
metavar=("MIN", "MAX"))
ap.add_argument("--only", default=",".join(ALL_KINDS),
help="comma-separated surfaces to sync: "
"posts,reels,stories,highlights. Use --only stories "
"for the cheap daily run.")
ap.add_argument("--min-interval", type=float, default=20.0, metavar="HOURS",
help="refuse to re-fetch a source touched more recently "
"than this (default 20h); the guard that makes a "
"restart cheap instead of a repeat")
ap.add_argument("--max-sources", type=int, default=0, metavar="N",
help="hard ceiling on sources touched in one run "
"(0 = no limit)")
ap.add_argument("--abort", type=int, default=0, metavar="N",
help="stop enumerating posts/reels after N consecutive "
"already-archived FILES (0 = walk everything, the "
"default). 50 is a safe routine value; it cuts the "
"per-run listing cost by roughly 85%%, at the price "
"of no longer noticing edited carousels")
ap.add_argument("--probe-ttl", type=float, default=24.0, metavar="HOURS",
help="reuse cached listing results younger than this")
ap.add_argument("--force", action="store_true",
help="ignore --min-interval and the probe cache")
mode = ap.add_mutually_exclusive_group()
mode.add_argument("--dry-run", action="store_true", default=True,
help="print the plan and the config; default")
mode.add_argument("--execute", action="store_true",
help="actually run gallery-dl")
args = ap.parse_args()
if not shutil.which("gallery-dl"):
print("gallery-dl not on PATH", file=sys.stderr)
return 2
if not shutil.which("rsync"):
print("rsync not on PATH", file=sys.stderr)
return 2
if args.post_url:
return run_post_urls(args)
if not args.index:
print("--index is required unless --post-url is given", file=sys.stderr)
return 2
index = ArchiveIndex(args.index)
names = index.profiles()
if args.urls_file:
if not args.urls_file.is_file():
print(f"urls file not found: {args.urls_file}", file=sys.stderr)
return 2
wanted = read_urls_file(args.urls_file)
if not wanted:
print(f"no usable profiles in {args.urls_file}", file=sys.stderr)
return 2
print(f"read {len(wanted)} profile(s) from {args.urls_file}")
selected = [Profile(p) for p in wanted]
elif args.profile:
for p in args.profile:
if p not in names:
print(f"note: {p} is not in the index yet; it will be created")
selected = [Profile(p) for p in args.profile]
else:
selected = [Profile(p) for p in sorted(names)]
config = build_config(args.rate, list(args.sleep_request),
list(args.sleep), args.abort)
args.staging.mkdir(parents=True, exist_ok=True)
# Deliberately a SIBLING of the staging directory, not inside it: staging is
# rsynced wholesale into the archive, and a dry run caught this file being
# published to the archive root.
config_path = args.staging.parent / f"{args.staging.name}.gdl-config.json"
kinds = {k.strip() for k in args.only.split(",") if k.strip()}
unknown = kinds - set(ALL_KINDS)
if unknown:
print(f"unknown surface(s): {', '.join(sorted(unknown))}", file=sys.stderr)
return 2
state_path = (args.archive_db.with_suffix(".state.json") if args.archive_db
else args.staging.parent / f"{args.staging.name}.state.json")
state = SyncState(state_path)
now = dt.datetime.now(dt.timezone.utc)
now_ts, stamp = now.timestamp(), now.isoformat()
min_interval = 0.0 if args.force else args.min_interval
plan: list[tuple[Profile, Source, bool]] = []
skipped = 0
for prof in selected:
for src in prof.sources(kinds):
fetch, seed, reason = plan_source(src, state, now_ts, min_interval)
if not fetch:
skipped += 1
print(f" skip {prof.user}/{src.kind}: {reason}")
continue
if args.max_sources and len(plan) >= args.max_sources:
skipped += 1
continue
plan.append((prof, src, seed))
print(f"profiles : {len(selected)}")
print(f"surfaces : {','.join(k for k in ALL_KINDS if k in kinds)}")
print(f"sources : {len(plan)} to sync, {skipped} skipped")
print(f"pacing : {args.sleep_request[0]}-{args.sleep_request[1]}s between "
f"requests, rate cap {args.rate}")
print(f"staging : {args.staging}")
print(f"publish : {args.publish}")
print()
if not args.execute:
for prof, src, seed in plan:
dest = src.directory or "(per-highlight)"
note = " [will seed]" if seed else ""
print(f" {prof.user:<20} {src.kind:<11} -> {dest}{note}")
print()
print(" " + " ".join(rsync_command(args.staging, args.publish, True)))
print("\ndry run; nothing fetched. pass --execute to run.")
return 0
config_path.write_text(json.dumps(config, indent=2))
probes = ProbeCache(state_path.with_suffix(".probes.json"),
0.0 if args.force else args.probe_ttl)
failures = 0
for prof, src, seed in plan:
print(f"==> {prof.user} / {src.kind}")
stage_dir = args.staging / (src.directory or ".")
stage_dir.mkdir(parents=True, exist_ok=True)
# Prime the skip-archive from what the archive already holds, so
# fetching into an empty staging directory pulls only what is missing.
# Done once per source, ever: afterwards the archive DB records
# everything gallery-dl has seen and no listing pass is needed.
if seed and args.archive_db:
try:
live = probes.get(src.url, now_ts)
if live is None:
live = probe_live(src, config_path, args.cookies)
probes.put(src.url, live, stamp)
probes.save()
else:
print(f" reusing {len(live)} cached listing items")
held = index_existing(index.listing(prof.user))
seeded = seed_archive_db(args.archive_db, held, live,
src.subcategory)
print(f" seeded {seeded} of {len(live)} live items")
state.mark_seeded(src.url, stamp)
state.save()
except subprocess.CalledProcessError as exc:
failures += 1
print(f" probe FAILED: {exc}", file=sys.stderr)
continue
cmd = gdl_command(src, args.staging, config_path, args.cookies,
args.archive_db)
result = subprocess.run(cmd)
if result.returncode != 0:
failures += 1
# Keep going: one private or renamed profile must not abort the run.
print(f" FAILED (exit {result.returncode})", file=sys.stderr)
else:
# Recorded even for an empty fetch: the request was still spent.
state.mark_fetched(src.url, stamp)
state.save()
# Publish once, at the end, so a partially-fetched profile never reaches
# the archive mid-run. Only ever adds -- see rsync_command.
print("\n==> publish")
if publish(args.staging, args.publish, dry_run=False) != 0:
failures += 1
print(f"\ndone; {failures} step(s) failed")
return 1 if failures else 0
if __name__ == "__main__":
sys.exit(main())