feat: drive gdl-sync from a file of profile URLs

Adds --urls-file so a run can reference a hand-maintained list rather
than repeating --profile, which is how this actually gets used: the
ARTMS accounts now live in artms_account_links.txt at the archive root.

The parser takes what a person would paste. Full URLs, scheme-less URLs
and bare usernames all work; blank lines and # comments are ignored and
duplicates dropped, so the list can be appended to carelessly. Lines that
are not profiles are rejected loudly rather than silently syncing
nothing: an Instagram post URL yields the segment "p", which would
otherwise be treated as a username and create a directory called "p".

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
2026-08-16 22:32:06 -04:00
co-authored by Claude Opus 5
parent 68568ef855
commit ee66b87bff
2 changed files with 70 additions and 6 deletions
Binary file not shown.
+70 -6
View File
@@ -6,9 +6,8 @@ The CLI replacement for the JDownloader2 workflow. See docs/gallery-dl.md for
the measurements behind every choice here — especially the safety model, which the measurements behind every choice here — especially the safety model, which
is the reason this script exists in this shape rather than a simpler one. is the reason this script exists in this shape rather than a simpler one.
STATUS: exercised end-to-end against `withaseul` (posts, reels, stories, STATUS: in use. `withaseul` was fetched and published to the live archive
highlights) publishing to a scratch directory. It has never written to the live (322 files added, nothing overwritten or deleted).
archive.
The fetch host needs no copy of the archive. It stages locally and rsyncs The fetch host needs no copy of the archive. It stages locally and rsyncs
afterwards; what it already holds is learned from a *file listing* alone afterwards; what it already holds is learned from a *file listing* alone
@@ -17,9 +16,13 @@ afterwards; what it already holds is learned from a *file listing* alone
Usage: Usage:
./scripts/gdl-sync.py --index https://instaarchive.ergosteur.com \\ ./scripts/gdl-sync.py --index https://instaarchive.ergosteur.com \\
--staging /var/tmp/gdl --publish user@host:/path/to/archives \\ --staging /var/tmp/gdl --publish user@host:/path/to/archives \\
--profile 0ct0ber19 --dry-run --urls-file artms_account_links.txt --dry-run
# ...then swap --dry-run for --execute. --index also accepts a local path. # ...then swap --dry-run for --execute. --index also accepts a local path,
# and --profile / --all work instead of --urls-file.
Always --dry-run first: it prints the plan, and the publish step it reports is
the one that would touch the archive.
Run it from the host whose public IP matches the browser the cookie came from; Run it from the host whose public IP matches the browser the cookie came from;
using the cookie from elsewhere is what session-hijack detection looks for. using the cookie from elsewhere is what session-hijack detection looks for.
@@ -114,6 +117,54 @@ def scan_archives(root: Path) -> dict[str, Profile]:
return profiles return profiles
RE_PROFILE_URL = re.compile(
r"^(?:https?://)?(?:www\.)?instagram\.com/(?P<user>[^/?#\s]+)/?", re.I)
# Path segments that are Instagram features, not profiles. A line like
# ".../p/ABC123/" names a post, and treating "p" as a username would silently
# sync nothing under a nonsense directory.
RESERVED_SEGMENTS = {
"p", "reel", "reels", "stories", "explore", "accounts", "direct",
"tv", "s", "invites", "challenge", "about", "developer",
}
def read_urls_file(path: Path) -> list[str]:
"""
Read profile URLs (or bare usernames) from a file, one per line.
Written for hand-maintained lists: blank lines are skipped, `#` starts a
comment, and either a full URL or a bare username works. Order is kept and
duplicates dropped, so a list can be appended to without care.
"""
users: list[str] = []
seen: set[str] = set()
for lineno, raw in enumerate(path.read_text().splitlines(), 1):
line = raw.split("#", 1)[0].strip()
if not line:
continue
m = RE_PROFILE_URL.match(line)
user = m.group("user") if m else line.strip("/")
if not user or "/" in user or " " in user:
print(f"{path}:{lineno}: cannot read a username from {raw.strip()!r}",
file=sys.stderr)
continue
if user.lower() in RESERVED_SEGMENTS:
print(f"{path}:{lineno}: {user!r} is an Instagram path, not a "
f"profile — skipping", file=sys.stderr)
continue
if user in seen:
continue
seen.add(user)
users.append(user)
return users
class ArchiveIndex: class ArchiveIndex:
""" """
What the archive already holds, as filenames only. What the archive already holds, as filenames only.
@@ -451,6 +502,9 @@ def main() -> int:
g.add_argument("--profile", action="append", default=[], g.add_argument("--profile", action="append", default=[],
help="profile to sync; repeatable") help="profile to sync; repeatable")
g.add_argument("--all", action="store_true", help="every profile on disk") g.add_argument("--all", action="store_true", help="every profile on disk")
g.add_argument("--urls-file", type=Path,
help="file of Instagram profile URLs or usernames, one per "
"line; # comments and blank lines allowed")
ap.add_argument("--cookies", default="chrome:/home/matt/.config/google-chrome-devtools", ap.add_argument("--cookies", default="chrome:/home/matt/.config/google-chrome-devtools",
help="gallery-dl --cookies-from-browser value") help="gallery-dl --cookies-from-browser value")
ap.add_argument("--archive-db", type=Path, default=None, ap.add_argument("--archive-db", type=Path, default=None,
@@ -478,7 +532,17 @@ def main() -> int:
index = ArchiveIndex(args.index) index = ArchiveIndex(args.index)
names = index.profiles() names = index.profiles()
if args.profile: if args.urls_file:
if not args.urls_file.is_file():
print(f"urls file not found: {args.urls_file}", file=sys.stderr)
return 2
wanted = read_urls_file(args.urls_file)
if not wanted:
print(f"no usable profiles in {args.urls_file}", file=sys.stderr)
return 2
print(f"read {len(wanted)} profile(s) from {args.urls_file}")
selected = [Profile(p) for p in wanted]
elif args.profile:
for p in args.profile: for p in args.profile:
if p not in names: if p not in names:
print(f"note: {p} is not in the index yet; it will be created") print(f"note: {p} is not in the index yet; it will be created")