feat: drive gdl-sync from a file of profile URLs
Adds --urls-file so a run can reference a hand-maintained list rather than repeating --profile, which is how this actually gets used: the ARTMS accounts now live in artms_account_links.txt at the archive root. The parser takes what a person would paste. Full URLs, scheme-less URLs and bare usernames all work; blank lines and # comments are ignored and duplicates dropped, so the list can be appended to carelessly. Lines that are not profiles are rejected loudly rather than silently syncing nothing: an Instagram post URL yields the segment "p", which would otherwise be treated as a username and create a directory called "p". Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
Binary file not shown.
+70
-6
@@ -6,9 +6,8 @@ The CLI replacement for the JDownloader2 workflow. See docs/gallery-dl.md for
|
|||||||
the measurements behind every choice here — especially the safety model, which
|
the measurements behind every choice here — especially the safety model, which
|
||||||
is the reason this script exists in this shape rather than a simpler one.
|
is the reason this script exists in this shape rather than a simpler one.
|
||||||
|
|
||||||
STATUS: exercised end-to-end against `withaseul` (posts, reels, stories,
|
STATUS: in use. `withaseul` was fetched and published to the live archive
|
||||||
highlights) publishing to a scratch directory. It has never written to the live
|
(322 files added, nothing overwritten or deleted).
|
||||||
archive.
|
|
||||||
|
|
||||||
The fetch host needs no copy of the archive. It stages locally and rsyncs
|
The fetch host needs no copy of the archive. It stages locally and rsyncs
|
||||||
afterwards; what it already holds is learned from a *file listing* alone
|
afterwards; what it already holds is learned from a *file listing* alone
|
||||||
@@ -17,9 +16,13 @@ afterwards; what it already holds is learned from a *file listing* alone
|
|||||||
Usage:
|
Usage:
|
||||||
./scripts/gdl-sync.py --index https://instaarchive.ergosteur.com \\
|
./scripts/gdl-sync.py --index https://instaarchive.ergosteur.com \\
|
||||||
--staging /var/tmp/gdl --publish user@host:/path/to/archives \\
|
--staging /var/tmp/gdl --publish user@host:/path/to/archives \\
|
||||||
--profile 0ct0ber19 --dry-run
|
--urls-file artms_account_links.txt --dry-run
|
||||||
|
|
||||||
# ...then swap --dry-run for --execute. --index also accepts a local path.
|
# ...then swap --dry-run for --execute. --index also accepts a local path,
|
||||||
|
# and --profile / --all work instead of --urls-file.
|
||||||
|
|
||||||
|
Always --dry-run first: it prints the plan, and the publish step it reports is
|
||||||
|
the one that would touch the archive.
|
||||||
|
|
||||||
Run it from the host whose public IP matches the browser the cookie came from;
|
Run it from the host whose public IP matches the browser the cookie came from;
|
||||||
using the cookie from elsewhere is what session-hijack detection looks for.
|
using the cookie from elsewhere is what session-hijack detection looks for.
|
||||||
@@ -114,6 +117,54 @@ def scan_archives(root: Path) -> dict[str, Profile]:
|
|||||||
return profiles
|
return profiles
|
||||||
|
|
||||||
|
|
||||||
|
RE_PROFILE_URL = re.compile(
|
||||||
|
r"^(?:https?://)?(?:www\.)?instagram\.com/(?P<user>[^/?#\s]+)/?", re.I)
|
||||||
|
|
||||||
|
# Path segments that are Instagram features, not profiles. A line like
|
||||||
|
# ".../p/ABC123/" names a post, and treating "p" as a username would silently
|
||||||
|
# sync nothing under a nonsense directory.
|
||||||
|
RESERVED_SEGMENTS = {
|
||||||
|
"p", "reel", "reels", "stories", "explore", "accounts", "direct",
|
||||||
|
"tv", "s", "invites", "challenge", "about", "developer",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def read_urls_file(path: Path) -> list[str]:
|
||||||
|
"""
|
||||||
|
Read profile URLs (or bare usernames) from a file, one per line.
|
||||||
|
|
||||||
|
Written for hand-maintained lists: blank lines are skipped, `#` starts a
|
||||||
|
comment, and either a full URL or a bare username works. Order is kept and
|
||||||
|
duplicates dropped, so a list can be appended to without care.
|
||||||
|
"""
|
||||||
|
users: list[str] = []
|
||||||
|
seen: set[str] = set()
|
||||||
|
|
||||||
|
for lineno, raw in enumerate(path.read_text().splitlines(), 1):
|
||||||
|
line = raw.split("#", 1)[0].strip()
|
||||||
|
if not line:
|
||||||
|
continue
|
||||||
|
|
||||||
|
m = RE_PROFILE_URL.match(line)
|
||||||
|
user = m.group("user") if m else line.strip("/")
|
||||||
|
|
||||||
|
if not user or "/" in user or " " in user:
|
||||||
|
print(f"{path}:{lineno}: cannot read a username from {raw.strip()!r}",
|
||||||
|
file=sys.stderr)
|
||||||
|
continue
|
||||||
|
if user.lower() in RESERVED_SEGMENTS:
|
||||||
|
print(f"{path}:{lineno}: {user!r} is an Instagram path, not a "
|
||||||
|
f"profile — skipping", file=sys.stderr)
|
||||||
|
continue
|
||||||
|
if user in seen:
|
||||||
|
continue
|
||||||
|
|
||||||
|
seen.add(user)
|
||||||
|
users.append(user)
|
||||||
|
|
||||||
|
return users
|
||||||
|
|
||||||
|
|
||||||
class ArchiveIndex:
|
class ArchiveIndex:
|
||||||
"""
|
"""
|
||||||
What the archive already holds, as filenames only.
|
What the archive already holds, as filenames only.
|
||||||
@@ -451,6 +502,9 @@ def main() -> int:
|
|||||||
g.add_argument("--profile", action="append", default=[],
|
g.add_argument("--profile", action="append", default=[],
|
||||||
help="profile to sync; repeatable")
|
help="profile to sync; repeatable")
|
||||||
g.add_argument("--all", action="store_true", help="every profile on disk")
|
g.add_argument("--all", action="store_true", help="every profile on disk")
|
||||||
|
g.add_argument("--urls-file", type=Path,
|
||||||
|
help="file of Instagram profile URLs or usernames, one per "
|
||||||
|
"line; # comments and blank lines allowed")
|
||||||
ap.add_argument("--cookies", default="chrome:/home/matt/.config/google-chrome-devtools",
|
ap.add_argument("--cookies", default="chrome:/home/matt/.config/google-chrome-devtools",
|
||||||
help="gallery-dl --cookies-from-browser value")
|
help="gallery-dl --cookies-from-browser value")
|
||||||
ap.add_argument("--archive-db", type=Path, default=None,
|
ap.add_argument("--archive-db", type=Path, default=None,
|
||||||
@@ -478,7 +532,17 @@ def main() -> int:
|
|||||||
|
|
||||||
index = ArchiveIndex(args.index)
|
index = ArchiveIndex(args.index)
|
||||||
names = index.profiles()
|
names = index.profiles()
|
||||||
if args.profile:
|
if args.urls_file:
|
||||||
|
if not args.urls_file.is_file():
|
||||||
|
print(f"urls file not found: {args.urls_file}", file=sys.stderr)
|
||||||
|
return 2
|
||||||
|
wanted = read_urls_file(args.urls_file)
|
||||||
|
if not wanted:
|
||||||
|
print(f"no usable profiles in {args.urls_file}", file=sys.stderr)
|
||||||
|
return 2
|
||||||
|
print(f"read {len(wanted)} profile(s) from {args.urls_file}")
|
||||||
|
selected = [Profile(p) for p in wanted]
|
||||||
|
elif args.profile:
|
||||||
for p in args.profile:
|
for p in args.profile:
|
||||||
if p not in names:
|
if p not in names:
|
||||||
print(f"note: {p} is not in the index yet; it will be created")
|
print(f"note: {p} is not in the index yet; it will be created")
|
||||||
|
|||||||
Reference in New Issue
Block a user