feat: scrape reels by scrolling the real page, since the API is blocked

gallery-dl's dedicated reels extractor POSTs to /api/v1/clips/user/,
which now 302-redirects for this account -- confirmed across multiple
profiles, hours apart, with a freshly-warmed session and a correct
X-IG-WWW-Claim header (ruled out as the cause). The reels tab itself
loads fine in a real, already-signed-in browser, so reels-scrape.py
drives that same Chrome via its loopback CDP port, scrolls the reels
tab like a person would, and scrapes /reel/<code>/ links out of the
rendered page instead of calling the blocked endpoint at all.

It only finds shortcodes -- deduped against the archive via the same
--index gdl-sync.py already uses -- and prints new post URLs. Feeding
many of those into gdl-sync.py needed two small additions: a
--post-urls-file so the list doesn't have to become a giant argv, and
inter-item pacing in run_post_urls (each --post-url was its own
subprocess with nothing pacing the gap between them).

Verified end to end against zindoriyam: 26 reels found, 16 already
archived, 10 new ones fetched and published cleanly.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_011qAds5qr7nZRq5R4yAuxUk
This commit is contained in:
2026-08-27 14:36:09 -04:00
co-authored by Claude Sonnet 5
parent eceee6ec02
commit 2f46123022
4 changed files with 327 additions and 3 deletions
+44 -3
View File
@@ -89,10 +89,12 @@ import argparse
import datetime as dt
import json
import os
import random
import re
import shutil
import subprocess
import sys
import time
from dataclasses import dataclass, field
from pathlib import Path
@@ -219,6 +221,23 @@ def read_urls_file(path: Path) -> list[str]:
return users
def read_post_urls_file(path: Path) -> list[str]:
"""
Read individual post/reel URLs from a file, one per line -- the output
format `reels-scrape.py` writes. Blank lines and `#` comments are skipped;
order is kept and duplicates dropped, same conventions as read_urls_file.
"""
urls: list[str] = []
seen: set[str] = set()
for raw in path.read_text().splitlines():
line = raw.split("#", 1)[0].strip()
if not line or line in seen:
continue
seen.add(line)
urls.append(line)
return urls
class ArchiveIndex:
"""
What the archive already holds, as filenames only.
@@ -736,7 +755,14 @@ def run_post_urls(args) -> int:
config_path.write_text(json.dumps(config, indent=2))
failures = 0
for src in sources:
for i, src in enumerate(sources):
# gallery-dl's own sleep-request only paces requests INSIDE one
# invocation; each URL here is its own subprocess, so back-to-back
# items would otherwise fire with no gap at all -- the same pacing
# applied here as between requests within a single fetch.
if i:
pause = random.uniform(*args.sleep_request)
time.sleep(pause)
print(f"==> {src.url}")
cmd = gdl_command(src, args.staging, config_path, args.cookies,
args.archive_db)
@@ -766,7 +792,8 @@ def main() -> int:
ap.add_argument("--index",
help="existing archive listing: a local root, or the "
"viewer's base URL (only a FILE LISTING is needed, "
"never the contents). Required unless --post-url")
"never the contents). Required unless --post-url/"
"--post-urls-file")
ap.add_argument("--publish", required=True,
help="rsync destination for fetched files; a local path or "
"user@host:/path")
@@ -785,6 +812,11 @@ def main() -> int:
"its owner's account like any other post; repeatable. "
"A one-off fetch outside the tracked profile list: no "
"archive-db seeding, no --min-interval floor")
g.add_argument("--post-urls-file", type=Path,
help="file of post/reel URLs, one per line; # comments and "
"blank lines allowed. Same handling as --post-url, "
"paced with --sleep-request between items. This is "
"the format reels-scrape.py writes")
ap.add_argument("--cookies", default="chrome:/home/matt/.config/google-chrome-devtools",
help="gallery-dl --cookies-from-browser value")
ap.add_argument("--archive-db", type=Path, default=None,
@@ -829,11 +861,20 @@ def main() -> int:
print("rsync not on PATH", file=sys.stderr)
return 2
if args.post_urls_file:
if not args.post_urls_file.is_file():
print(f"post-urls file not found: {args.post_urls_file}", file=sys.stderr)
return 2
args.post_url = read_post_urls_file(args.post_urls_file)
if not args.post_url:
print(f"no usable URLs in {args.post_urls_file}", file=sys.stderr)
return 2
if args.post_url:
return run_post_urls(args)
if not args.index:
print("--index is required unless --post-url is given", file=sys.stderr)
print("--index is required unless --post-url/--post-urls-file is given",
file=sys.stderr)
return 2
index = ArchiveIndex(args.index)
+225
View File
@@ -0,0 +1,225 @@
#!/usr/bin/env python3
"""
Find a profile's reels by scrolling the real page, not calling the API.
gallery-dl's dedicated reels extractor POSTs to /api/v1/clips/user/, which
Instagram now 302-redirects to the home page for this account -- confirmed on
2026-08-26/27 across multiple profiles, hours apart, with a freshly-warmed
session and a correct X-IG-WWW-Claim header. The reels tab itself loads fine
in a real browser, so this drives the SAME logged-in Chrome (already exposed
on loopback CDP for MCP automation -- see TOOLING.md) via the DevTools
protocol, scrolls it like a person would, and scrapes `/reel/<code>/` links
out of the rendered page instead.
This only finds shortcodes; it never downloads anything itself. What comes
out the other end is deduped against the archive (via the same --index used
elsewhere) and printed as plain post URLs -- feed them to gdl-sync.py:
./scripts/reels-scrape.py --profile someuser \\
--index https://instaarchive.ergosteur.com > /tmp/someuser-reels.txt
./scripts/gdl-sync.py --publish user@host:/path --staging /var/tmp/gdl \\
--post-urls-file /tmp/someuser-reels.txt \\
--sleep-request 12 20 --sleep 5 10 --rate 500K --execute
Requires the `websocket-client` package (only imported inside scrape_reels,
so everything else here stays importable -- and testable -- without it).
"""
from __future__ import annotations
import argparse
import itertools
import json
import random
import re
import sys
import time
from pathlib import Path
from urllib.request import urlopen
RE_REEL_HREF = re.compile(r"/reel/([^/?#]+)")
# The reels tab has finished loading once a scroll round finds no NEW
# shortcodes this many times in a row -- lazy-loaded feeds sometimes stall
# for a round or two before producing more, so one dry round is not enough.
DEFAULT_MAX_IDLE_ROUNDS = 3
DEFAULT_MAX_SCROLLS = 200
DEFAULT_SCROLL_PAUSE = (2.0, 3.5)
SCRAPE_JS = (
"(() => {"
"window.scrollTo(0, document.body.scrollHeight);"
"return Array.from(document.querySelectorAll('a[href*=\"/reel/\"]'))"
".map(a => a.getAttribute('href'));"
"})()"
)
def extract_shortcodes(hrefs: list[str]) -> list[str]:
codes = []
for href in hrefs:
if m := RE_REEL_HREF.search(href):
codes.append(m.group(1))
return codes
class CDPError(RuntimeError):
pass
class CDP:
"""
A deliberately minimal synchronous DevTools Protocol client: one request
in flight at a time, which is all a linear scroll-and-scrape loop needs.
Anything fancier (concurrent requests, event subscriptions) is scope this
script has no reason to carry.
"""
def __init__(self, ws_url: str, timeout: float = 30.0):
import websocket # local: keep this importable without the package
self.ws = websocket.create_connection(ws_url, timeout=timeout)
self._ids = itertools.count(1)
def send(self, method: str, params: dict | None = None,
session_id: str | None = None) -> dict:
msg_id = next(self._ids)
payload = {"id": msg_id, "method": method, "params": params or {}}
if session_id:
payload["sessionId"] = session_id
self.ws.send(json.dumps(payload))
while True:
msg = json.loads(self.ws.recv())
if msg.get("id") != msg_id:
continue # an event notification, not our reply -- ignore
if "error" in msg:
raise CDPError(f"{method}: {msg['error']}")
return msg.get("result", {})
def close(self) -> None:
self.ws.close()
def scrape_reels(user: str, cdp_port: int = 9222,
scroll_pause: tuple[float, float] = DEFAULT_SCROLL_PAUSE,
max_idle_rounds: int = DEFAULT_MAX_IDLE_ROUNDS,
max_scrolls: int = DEFAULT_MAX_SCROLLS,
log=lambda msg: None) -> list[str]:
"""
Open the profile's reels tab in a NEW tab of the already-signed-in Chrome,
scroll it to the bottom repeatedly, and collect every unique `/reel/`
shortcode that appears. Closes the tab when done either way.
"""
version = json.loads(urlopen(f"http://localhost:{cdp_port}/json/version",
timeout=10).read())
browser = CDP(version["webSocketDebuggerUrl"])
target_id = None
try:
target = browser.send("Target.createTarget", {
"url": f"https://www.instagram.com/{user}/reels/"})
target_id = target["targetId"]
attach = browser.send("Target.attachToTarget", {
"targetId": target_id, "flatten": True})
session_id = attach["sessionId"]
browser.send("Page.enable", session_id=session_id)
browser.send("Runtime.enable", session_id=session_id)
time.sleep(4.0) # initial page load, before the first scroll
seen: set[str] = set()
idle = 0
for i in range(max_scrolls):
result = browser.send(
"Runtime.evaluate",
{"expression": SCRAPE_JS, "returnByValue": True},
session_id=session_id)
hrefs = result.get("result", {}).get("value") or []
codes = extract_shortcodes(hrefs)
new = [c for c in codes if c not in seen]
seen.update(new)
log(f" scroll {i + 1}: {len(seen)} unique reels so far (+{len(new)})")
if new:
idle = 0
else:
idle += 1
if idle >= max_idle_rounds:
break
time.sleep(random.uniform(*scroll_pause))
return sorted(seen)
finally:
if target_id:
try:
browser.send("Target.closeTarget", {"targetId": target_id})
except CDPError:
pass # best-effort cleanup; a leftover tab is harmless
browser.close()
def main() -> int:
ap = argparse.ArgumentParser(description=__doc__,
formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument("--profile", required=True,
help="Instagram username to scrape reels for")
ap.add_argument("--index", required=True,
help="existing archive listing, same as gdl-sync.py's "
"--index: a local root, or the viewer's base URL. "
"Used only to dedupe -- already-archived shortcodes "
"are dropped before anything is printed")
ap.add_argument("--cdp-port", type=int, default=9222,
help="Chrome's loopback DevTools port (see TOOLING.md)")
ap.add_argument("--scroll-pause", nargs=2, type=float,
default=list(DEFAULT_SCROLL_PAUSE), metavar=("MIN", "MAX"),
help="random pause between scrolls, seconds")
ap.add_argument("--max-idle-rounds", type=int, default=DEFAULT_MAX_IDLE_ROUNDS,
help="stop after this many consecutive scrolls with no "
"new reels")
ap.add_argument("--max-scrolls", type=int, default=DEFAULT_MAX_SCROLLS,
help="hard ceiling on scroll rounds, in case a page never "
"goes idle")
ap.add_argument("--out", type=Path,
help="write new reel URLs here, one per line (default: "
"stdout)")
args = ap.parse_args()
import importlib.util
spec = importlib.util.spec_from_file_location(
"gdl_sync", Path(__file__).with_name("gdl-sync.py"))
gdl = importlib.util.module_from_spec(spec)
sys.modules["gdl_sync"] = gdl
spec.loader.exec_module(gdl)
index = gdl.ArchiveIndex(args.index)
have = {code for code, _ in gdl.index_existing(index.listing(args.profile))}
print(f"archive already holds {len(have)} shortcode(s) for {args.profile}",
file=sys.stderr)
print(f"scraping https://www.instagram.com/{args.profile}/reels/ ...",
file=sys.stderr)
codes = scrape_reels(args.profile, cdp_port=args.cdp_port,
scroll_pause=tuple(args.scroll_pause),
max_idle_rounds=args.max_idle_rounds,
max_scrolls=args.max_scrolls,
log=lambda msg: print(msg, file=sys.stderr))
new_codes = [c for c in codes if c not in have]
print(f"found {len(codes)} reel(s) on the page, {len(codes) - len(new_codes)} "
f"already archived, {len(new_codes)} new", file=sys.stderr)
urls = [f"https://www.instagram.com/{args.profile}/reel/{c}/"
for c in new_codes]
text = "\n".join(urls)
if args.out:
args.out.write_text(text + ("\n" if text else ""))
print(f"wrote {len(urls)} URL(s) to {args.out}", file=sys.stderr)
else:
if text:
print(text)
return 0
if __name__ == "__main__":
sys.exit(main())
+18
View File
@@ -250,5 +250,23 @@ class UrlsFile(unittest.TestCase):
self.assertEqual(gdl.read_urls_file(p), ["a", "b", "c", "d", "e"])
class PostUrlsFile(unittest.TestCase):
"""The format reels-scrape.py writes: whole URLs, not usernames."""
def test_skips_comments_blanks_and_duplicates(self):
with tempfile.TemporaryDirectory() as d:
p = Path(d) / "reels.txt"
p.write_text(
"# scraped 2026-08-27\n"
"https://www.instagram.com/u/reel/AAA/\n"
"\n"
"https://www.instagram.com/u/reel/BBB/\n"
"https://www.instagram.com/u/reel/AAA/\n") # duplicate
self.assertEqual(gdl.read_post_urls_file(p), [
"https://www.instagram.com/u/reel/AAA/",
"https://www.instagram.com/u/reel/BBB/",
])
if __name__ == "__main__":
unittest.main(verbosity=2)
+40
View File
@@ -0,0 +1,40 @@
#!/usr/bin/env python3
"""
Tests for the pure parts of reels-scrape.py -- everything except the actual
CDP session, which needs a live signed-in Chrome and is exercised by hand.
python3 -m unittest discover -s scripts -p 'test_*.py'
"""
import importlib.util
import sys
import unittest
from pathlib import Path
_spec = importlib.util.spec_from_file_location(
"reels_scrape", Path(__file__).with_name("reels-scrape.py"))
scrape = importlib.util.module_from_spec(_spec)
sys.modules["reels_scrape"] = scrape
_spec.loader.exec_module(scrape)
class ExtractShortcodes(unittest.TestCase):
def test_reads_relative_and_absolute_hrefs(self):
hrefs = [
"/someuser/reel/ABC123/",
"https://www.instagram.com/someuser/reel/DEF456/",
"/reel/GHI789/?img_index=1",
]
self.assertEqual(scrape.extract_shortcodes(hrefs),
["ABC123", "DEF456", "GHI789"])
def test_ignores_non_reel_links(self):
hrefs = ["/someuser/", "/someuser/p/ABC123/", "/explore/tags/foo/"]
self.assertEqual(scrape.extract_shortcodes(hrefs), [])
def test_empty_input(self):
self.assertEqual(scrape.extract_shortcodes([]), [])
if __name__ == "__main__":
unittest.main(verbosity=2)