/** * Generate JDownloader2 .crawljob files for the archives on disk. * * The manual flow is: paste a profile URL into JDownloader, paste the /reels * URL separately (the profile page misses some reels), and set the output * folder by hand — times however many profiles you keep. This emits one * crawljob per source with the folder already pointed at the right directory, * so JDownloader's folder-watch picks the whole batch up at once. * * Profiles and their sidecar directories are derived with the same grouping * logic the server uses, so the output folders always match what the viewer * expects to find. * * Only posts and reels are emitted. Story and highlight URLs can't be rebuilt * from a directory name — highlights need their numeric id and stories expire — * so those stay manual. * * Crawljob format verified against JDownloader's own docs for the extension: * src/org/jdownloader/extensions/folderwatchV2/explain.txt. JDownloader * develops on SVN; read it via the daily mirror at * https://github.com/mycodedoesnotcompile2/jdownloader_mirror (svn_trunk/), * not one of the abandoned GitHub copies — several are a decade stale. * * Entries are separated by `->NEW ENTRY<-` and any property may be omitted. * There is also a `setBeforePackagizerEnabled` companion to * `overwritePackagizerEnabled`, if the Packagizer ever needs to see these * values before they're applied. * * Usage: * npx tsx scripts/jd2-sync.ts --archives [options] * * --archives Archive root to scan (default: $ARCHIVES_DIR) * --out JDownloader folder-watch directory to write into * --download-base Root path as *JDownloader* sees it, when it runs on * a different machine than this script (e.g. a mapped * drive). Defaults to --archives. * --user Only this profile (repeatable) * --skip Never emit jobs for this directory (repeatable). * Also read from a `.jd2ignore` file in the archive * root, one name per line. * --chunks Connections per file (default 1: multi-chunk ranged * requests are the one CDN pattern that doesn't look * like a browser) * --auto-start Start downloads immediately instead of parking them * in the LinkGrabber for review * --all-reels Emit a reels job even where no reels directory * exists yet * --dry-run Print the crawljob instead of writing it */ import fs from 'fs'; import path from 'path'; import { groupArchiveDirectories, ArchiveSource } from '../src/lib/archive-grouping.js'; interface Options { archives: string; out: string | null; downloadBase: string; users: string[]; skip: Set; chunks: number; autoStart: boolean; allReels: boolean; dryRun: boolean; } const parseArgs = (argv: string[]): Options => { const opts: Options = { archives: process.env.ARCHIVES_DIR ?? '', out: null, downloadBase: '', users: [], skip: new Set(), chunks: 1, autoStart: false, allReels: false, dryRun: false, }; for (let i = 0; i < argv.length; i++) { const arg = argv[i]; const next = () => argv[++i]; switch (arg) { case '--archives': opts.archives = path.resolve(next()); break; case '--out': opts.out = path.resolve(next()); break; case '--download-base': opts.downloadBase = next(); break; case '--user': opts.users.push(next()); break; case '--skip': opts.skip.add(next()); break; case '--chunks': opts.chunks = parseInt(next(), 10); break; case '--auto-start': opts.autoStart = true; break; case '--all-reels': opts.allReels = true; break; case '--dry-run': opts.dryRun = true; break; case '--help': case '-h': printUsage(); process.exit(0); default: console.error(`Unknown argument: ${arg}`); process.exit(1); } } if (!opts.archives) { console.error('No archive root. Pass --archives or set ARCHIVES_DIR.'); process.exit(1); } if (!opts.downloadBase) opts.downloadBase = opts.archives; if (!opts.out && !opts.dryRun) { console.error('No destination. Pass --out , or --dry-run to preview.'); process.exit(1); } return opts; }; const printUsage = () => { const header = readHeaderComment(); console.log(header); }; /** Print the usage block from this file's own header comment. */ const readHeaderComment = () => { try { const self = fs.readFileSync(new URL(import.meta.url), 'utf8'); const usage = self.slice(self.indexOf(' * Usage:'), self.indexOf(' */')); return usage.split('\n').map(l => l.replace(/^ \* ?/, '')).join('\n'); } catch { return 'See the comment at the top of scripts/jd2-sync.ts'; } }; /** * JDownloader escapes nothing in crawljob values, so a stray newline would * silently split a property. Paths with spaces are fine as-is. */ const sanitise = (value: string) => value.replace(/[\r\n]+/g, ' ').trim(); /** * Instagram usernames are 1–30 characters of letters, digits, dots and * underscores. Archive roots also collect directories that aren't profiles at * all — tool output, exports from other services — and pointing a crawl at * those spends requests on instagram.com to be told the profile doesn't exist. * That's the exact traffic worth not spending. */ const USERNAME_RE = /^[A-Za-z0-9._]{1,30}$/; /** Directory names to skip, from `.jd2ignore` in the archive root. */ const readIgnoreFile = (archives: string): string[] => { try { return fs.readFileSync(path.join(archives, '.jd2ignore'), 'utf8') .split('\n').map(l => l.trim()).filter(l => l && !l.startsWith('#')); } catch { return []; } }; interface Job { user: string; kind: 'posts' | 'reels'; url: string; packageName: string; downloadFolder: string; fileCount: number | null; } const buildJobs = (opts: Options): Job[] => { const dirNames = fs.readdirSync(opts.archives, { withFileTypes: true }) .filter(e => e.isDirectory() && !/^[.@_]/.test(e.name)) .map(e => e.name); const groups = groupArchiveDirectories(dirNames); const jobs: Job[] = []; const skipped: string[] = []; for (const name of readIgnoreFile(opts.archives)) opts.skip.add(name); const countFiles = (dir: string): number | null => { try { return fs.readdirSync(path.join(opts.archives, dir)).length; } catch { return null; } }; // JDownloader must be given the path *it* can see, which differs from the // scan path whenever the archive lives on a share. const downloadFolderFor = (dir: string) => opts.downloadBase.includes('\\') ? `${opts.downloadBase.replace(/\\$/, '')}\\${dir}` : path.posix.join(opts.downloadBase, dir); for (const [user, sources] of [...groups].sort(([a], [b]) => a.localeCompare(b))) { if (opts.users.length && !opts.users.includes(user)) continue; if (opts.skip.has(user)) { skipped.push(`${user} (ignored)`); continue; } if (!USERNAME_RE.test(user)) { skipped.push(`${user} (not a username)`); continue; } const has = (kind: ArchiveSource['kind']) => sources.find(s => s.kind === kind); const base = has('posts'); if (!base) continue; // sidecar-only group: nothing sensible to point a URL at jobs.push({ user, kind: 'posts', url: `https://www.instagram.com/${encodeURIComponent(user)}/`, packageName: base.dir, downloadFolder: downloadFolderFor(base.dir), fileCount: countFiles(base.dir), }); const reels = has('reels'); if (reels || opts.allReels) { const dir = reels?.dir ?? `${user} - reels`; jobs.push({ user, kind: 'reels', url: `https://www.instagram.com/${encodeURIComponent(user)}/reels/`, packageName: dir, downloadFolder: downloadFolderFor(dir), fileCount: reels ? countFiles(dir) : null, }); } } if (skipped.length) { console.error(`Skipped ${skipped.length} director${skipped.length === 1 ? 'y' : 'ies'}:`); for (const s of skipped) console.error(` - ${s}`); console.error(''); } return jobs; }; const renderCrawljob = (jobs: Job[], opts: Options): string => jobs.map(job => [ `text=${sanitise(job.url)}`, `packageName=${sanitise(job.packageName)}`, `downloadFolder=${sanitise(job.downloadFolder)}`, `chunks=${opts.chunks}`, // Without this a Packagizer rule can override downloadFolder and scatter // files away from the directory the viewer reads. 'overwritePackagizerEnabled=TRUE', `autoStart=${opts.autoStart ? 'TRUE' : 'FALSE'}`, `autoConfirm=${opts.autoStart ? 'TRUE' : 'FALSE'}`, 'enabled=TRUE', `comment=instaarchive jd2-sync (${job.kind})`, ].join('\n')).join('\n->NEW ENTRY<-\n'); const main = () => { const opts = parseArgs(process.argv.slice(2)); const jobs = buildJobs(opts); if (!jobs.length) { console.error('No profiles matched.'); process.exit(1); } console.error(`Archive root : ${opts.archives}`); console.error(`JD sees root : ${opts.downloadBase}`); console.error(`Jobs : ${jobs.length} (${new Set(jobs.map(j => j.user)).size} profiles)\n`); for (const job of jobs) { const count = job.fileCount === null ? 'new' : `${job.fileCount} files`; console.error(` ${job.kind.padEnd(5)} ${job.user.padEnd(24)} -> ${job.packageName} (${count})`); } console.error(''); const body = renderCrawljob(jobs, opts); if (opts.dryRun || !opts.out) { console.log(body); return; } fs.mkdirSync(opts.out, { recursive: true }); const file = path.join(opts.out, `instaarchive-${new Date().toISOString().replace(/[:.]/g, '-')}.crawljob`); fs.writeFileSync(file, body, 'utf8'); console.error(`Wrote ${file}`); console.error(opts.autoStart ? 'Downloads will start automatically.' : 'Links land in the LinkGrabber for review; start them when ready.'); }; main();