#!/usr/bin/env python3 """ Fetch Instagram profiles into the archive layout using gallery-dl. The CLI replacement for the JDownloader2 workflow. See docs/gallery-dl.md for the measurements behind every choice here — especially the safety model, which is the reason this script exists in this shape rather than a simpler one. STATUS: skeleton. The config generation and planning are complete and tested; `--execute` is deliberately gated behind an explicit flag and has not been run against the live archive. Usage: ./scripts/gdl-sync.py --archives --profile 0ct0ber19 --dry-run ./scripts/gdl-sync.py --archives --all --dry-run ./scripts/gdl-sync.py --archives --profile 0ct0ber19 --execute Run it from the host whose public IP matches the browser the cookie came from. """ from __future__ import annotations import argparse import json import os import re import shutil import subprocess import sys from dataclasses import dataclass, field from pathlib import Path # -------------------------------------------------------------------------- # Archive layout # -------------------------------------------------------------------------- # Mirrors src/lib/archive-grouping.ts. Instagram usernames cannot contain # spaces, which is what makes the username separable from a highlight title. RE_HIGHLIGHT = re.compile(r"^story highlights - ([^ ]+) - (.+)$") RE_STORIES = re.compile(r"^story - ([^ ]+)$") RE_REELS = re.compile(r"^([^ ]+) - reels$") DATE_FMT = "{date:Olocal/%Y-%m-%d}" """Local-time date. JD2 stamped US Eastern, NOT UTC (0/212 mismatches vs 19 for UTC). `Olocal` is DST-aware per timestamp. The trailing separator must be omitted or it lands in the strftime format and sanitises to an underscore.""" POST_STEM = DATE_FMT + "_{username} - {post_shortcode}" ITEM_STEM = DATE_FMT + "_{username} - {shortcode}" @dataclass class Source: """One gallery-dl invocation: a URL fetched into a specific directory.""" kind: str # posts | reels | stories | highlights url: str directory: str # relative to the archives root subcategory: str # gallery-dl config key title: str | None = None # highlight title, when known @dataclass class Profile: user: str existing: dict[str, str] = field(default_factory=dict) # kind -> dirname def sources(self, include_stories: bool) -> list[Source]: u = self.user base = f"https://www.instagram.com/{u}" out = [ Source("posts", f"{base}/posts/", u, "posts"), Source("reels", f"{base}/reels/", f"{u} - reels", "reels"), ] if include_stories: # Stories expire after 24h, so these can only ever be captured # live. There is no backfill and no re-fetch. out.append(Source( "stories", f"https://www.instagram.com/stories/{u}/", f"story - {u}", "stories")) # Highlight directories embed the title, which gallery-dl only # learns mid-extraction -- so this one source fans out into many # directories and is handled with a directory format string. out.append(Source( "highlights", f"{base}/highlights", "", "highlights")) return out def scan_archives(root: Path) -> dict[str, Profile]: """Group existing directories into profiles, as the server does.""" profiles: dict[str, Profile] = {} def get(user: str) -> Profile: return profiles.setdefault(user, Profile(user)) for entry in sorted(os.listdir(root)): if not (root / entry).is_dir() or entry.startswith("."): continue if m := RE_HIGHLIGHT.match(entry): get(m.group(1)).existing.setdefault("highlights", entry) elif m := RE_STORIES.match(entry): get(m.group(1)).existing["stories"] = entry elif m := RE_REELS.match(entry): get(m.group(1)).existing["reels"] = entry else: get(entry).existing["posts"] = entry return profiles # -------------------------------------------------------------------------- # gallery-dl configuration # -------------------------------------------------------------------------- def build_config(rate: str, sleep_request: list[float], sleep: list[float]) -> dict: """ The config is generated rather than checked in so the safety-critical options cannot drift out of sync with the docs. `api: rest` is the single most important line in this file. The graphql backend issues one request PER POST for every video and carousel, which is the pattern that got this account banned once already. """ caption_pp = { "name": "metadata", "event": "post", "mode": "custom", "content-format": "{description}", "extension": "txt", # JD2 wrote no .txt when the caption was empty; "empty": false (the # default) reproduces that. } meta_pp = { "name": "metadata", "event": "post", "mode": "json", # `include`, NOT `fields` -- `fields` applies to mode:custom and # silently does nothing here, dumping audio_user blobs that contain # unrelated users' profile picture URLs. "include": [ "post_shortcode", "post_id", "type", "date", "post_date", "username", "fullname", "owner_id", "description", "count", "likes", "post_url", "sidecar_shortcode", ], } def post_like(stem: str) -> dict: """Naming for surfaces whose unit is a post (posts, reels).""" return { # `sidecar_shortcode` is set only for carousels, so it is the # carousel discriminator. First matching condition wins. "filename": { "sidecar_shortcode and count >= 10": stem + " - {num:02}.{extension}", "sidecar_shortcode": stem + " - {num}.{extension}", "": stem + ".{extension}", }, "postprocessors": [ {**caption_pp, "filename": stem + ".txt"}, {**meta_pp, "filename": stem + ".json"}, ], } def item_like(stem: str) -> dict: """ Naming for surfaces whose unit is an item inside a reel (stories, highlights). `{shortcode}` is per item; `{post_shortcode}` is the reel's id and is shared by every item in it. """ return { "filename": stem + ".{extension}", "postprocessors": [{**meta_pp, "filename": stem + ".json"}], } return { "extractor": { "base-directory": ".", "instagram": { "api": "rest", # never "graphql" -- see docstring "sleep-request": sleep_request, "sleep": sleep, "videos": True, "include": "", # never "all"; sources are explicit # Directory is forced per-invocation with -D, because a reels # tab returns collab reels owned by OTHER accounts and # {username} would scatter them into the wrong profile. "directory": [], "posts": post_like(POST_STEM), "reels": post_like(POST_STEM), "stories": item_like(ITEM_STEM), "highlights": { **item_like(ITEM_STEM), # The only surface that must derive its own directory, # since the title is not known until extraction. "directory": ["story highlights - {username} - {highlight_title}"], }, }, }, "downloader": {"http": {"rate": rate}}, "output": {"mode": "null"}, } # -------------------------------------------------------------------------- # Planning and execution # -------------------------------------------------------------------------- def gdl_command(src: Source, root: Path, config: Path, cookies: str, archive_db: Path | None) -> list[str]: cmd = [ "gallery-dl", "--config", str(config), "--cookies-from-browser", cookies, ] if archive_db: # gallery-dl's skip-archive. Seed it before the first real run or the # whole 110k-file tree is re-downloaded; see --seed-archive. cmd += ["--download-archive", str(archive_db)] if src.subcategory != "highlights": cmd += ["--destination", str(root / src.directory)] else: cmd += ["--destination", str(root)] cmd.append(src.url) return cmd # gallery-dl keys its skip-archive on `archive_prefix + archive_fmt`, which for # this extractor is the literal "instagram" followed by the per-media numeric # pk (`instagram.py:25`, `job.py:713-719`). Verified against a real run: a # 3-image carousel produced 3 rows, one per item. ARCHIVE_KEY = "instagram{}".format ARCHIVE_SCHEMA = "CREATE TABLE IF NOT EXISTS archive (entry TEXT PRIMARY KEY)" RE_ARCHIVED = re.compile( r"^(\d{4}-\d{2}-\d{2})_(.+?) - ([A-Za-z0-9_-]+?)(?: - (\d+))?\.(\w+)$") NON_MEDIA = {"txt", "json"} def index_existing(listing: list[str]) -> set[tuple[str, int]]: """ Reduce a flat list of filenames to the (shortcode, index) pairs already held. Only names matter — never the bytes — which is what lets the sync run on a host that has no copy of the archive. """ have: set[tuple[str, int]] = set() for name in listing: m = RE_ARCHIVED.match(name.rsplit("/", 1)[-1]) if not m or m.group(5).lower() in NON_MEDIA: continue # An absent index means a single-media post, which is index 1 — the # same normalisation the viewer's EXPORT_RE applies. have.add((m.group(3), int(m.group(4) or 1))) return have def live_key(item: dict, kind: str) -> tuple[str, int]: """ The (shortcode, index) a live item *would* be filed under, mirroring the filename template exactly. The two surfaces disagree about which shortcode identifies a file, and getting this wrong silently seeds almost nothing: posts/reels filed under {post_shortcode} — for a carousel, each child item ALSO has its own `shortcode`, which is not what appears in the filename. stories/highlights filed under the per-item {shortcode}, because `post_shortcode` there is the containing reel's id and is shared by every item in it. """ if kind in ("stories", "highlights"): return (item.get("shortcode"), 1) return (item.get("post_shortcode"), item.get("num")) def seed_archive_db(db: Path, existing: set[tuple[str, int]], live: list[dict], kind: str) -> int: """ Mark everything already held as downloaded, so a fetch into an empty directory pulls only what is missing. `live` is the metadata of one listing pass — the pass we have to make anyway — each entry carrying at least `media_id` plus the shortcode fields `live_key` needs. Seeding costs no additional Instagram requests, and needs only a *listing* of the archive, never its contents. """ import sqlite3 db.parent.mkdir(parents=True, exist_ok=True) con = sqlite3.connect(db) con.execute(ARCHIVE_SCHEMA) rows = [ (ARCHIVE_KEY(item["media_id"]),) for item in live if live_key(item, kind) in existing ] con.executemany("INSERT OR IGNORE INTO archive (entry) VALUES (?)", rows) con.commit() con.close() return len(rows) def probe_live(src: Source, config: Path, cookies: str) -> list[dict]: """ One metadata-only listing pass. `sleep` is forced to 0 because it otherwise applies per *file* even with no download — 2275 files at 1-3s each is over an hour for a single profile. """ out = subprocess.run( ["gallery-dl", "-j", "--config", str(config), "--cookies-from-browser", cookies, "-o", "sleep=0", src.url], capture_output=True, text=True, check=True, ) items: list[dict] = [] def walk(node): if isinstance(node, dict): if "media_id" in node and "shortcode" in node: items.append(node) for value in node.values(): walk(value) elif isinstance(node, list): for value in node: walk(value) walk(json.loads(out.stdout)) return items def main() -> int: ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) ap.add_argument("--archives", required=True, type=Path, help="archive root (one directory per profile plus sidecars)") g = ap.add_mutually_exclusive_group(required=True) g.add_argument("--profile", action="append", default=[], help="profile to sync; repeatable") g.add_argument("--all", action="store_true", help="every profile on disk") ap.add_argument("--cookies", default="chrome:/home/matt/.config/google-chrome-devtools", help="gallery-dl --cookies-from-browser value") ap.add_argument("--archive-db", type=Path, default=None, help="gallery-dl skip-archive sqlite path") ap.add_argument("--rate", default="3M", help="per-download rate cap") ap.add_argument("--sleep-request", nargs=2, type=float, default=[4.0, 7.0], metavar=("MIN", "MAX")) ap.add_argument("--sleep", nargs=2, type=float, default=[1.0, 3.0], metavar=("MIN", "MAX")) ap.add_argument("--no-stories", action="store_true", help="skip stories and highlights (posts and reels only)") mode = ap.add_mutually_exclusive_group() mode.add_argument("--dry-run", action="store_true", default=True, help="print the plan and the config; default") mode.add_argument("--execute", action="store_true", help="actually run gallery-dl") args = ap.parse_args() if not args.archives.is_dir(): print(f"archives root not found: {args.archives}", file=sys.stderr) return 2 if not shutil.which("gallery-dl"): print("gallery-dl not on PATH", file=sys.stderr) return 2 profiles = scan_archives(args.archives) if args.profile: missing = [p for p in args.profile if p not in profiles] for p in missing: print(f"note: {p} has no directory yet; it will be created") profiles.setdefault(p, Profile(p)) selected = [profiles[p] for p in args.profile] else: selected = list(profiles.values()) config = build_config(args.rate, list(args.sleep_request), list(args.sleep)) config_path = args.archives / ".gdl-sync.config.json" plan: list[tuple[Profile, Source]] = [ (prof, src) for prof in selected for src in prof.sources(include_stories=not args.no_stories) ] print(f"profiles : {len(selected)}") print(f"sources : {len(plan)}") print(f"pacing : {args.sleep_request[0]}-{args.sleep_request[1]}s between " f"requests, rate cap {args.rate}") print() if not args.execute: print(json.dumps(config, indent=2)) print() for prof, src in plan: dest = src.directory or "(per-highlight)" print(f" {prof.user:<20} {src.kind:<11} -> {dest}") print("\ndry run; nothing fetched. pass --execute to run.") return 0 config_path.write_text(json.dumps(config, indent=2)) failures = 0 for prof, src in plan: if src.subcategory != "highlights": (args.archives / src.directory).mkdir(parents=True, exist_ok=True) cmd = gdl_command(src, args.archives, config_path, args.cookies, args.archive_db) print(f"==> {prof.user} / {src.kind}") result = subprocess.run(cmd) if result.returncode != 0: failures += 1 # Keep going: one private/renamed profile must not abort the run. print(f" FAILED (exit {result.returncode})", file=sys.stderr) # A sync NEVER deletes. The archive deliberately outlives Instagram -- # 2 posts in 0ct0ber19 exist only here now. print(f"\ndone; {failures} source(s) failed") return 1 if failures else 0 if __name__ == "__main__": sys.exit(main())