Files
instaarchive-viewer/src/lib/archive-patterns.test.ts
T
ergosteurandClaude Opus 5 89bd5346db docs: design a gallery-dl replacement for the JDownloader fetcher
Every claim in docs/gallery-dl.md was measured against the live site and
the archive rather than taken from documentation, because two of the
assumptions turned out to be wrong.

The safety model is the reason the config looks the way it does.
gallery-dl has two API backends: the graphql one issues a request PER
POST for every video and carousel -- the pattern that got this account
banned via Instaloader -- while the default rest one paginates listings
at 30-50 items and carries carousel_media, video_versions and
product_type inline. A 300-post profile costs ~10 requests.

Findings worth recording:

- JD2 stamped filenames in desktop LOCAL time (US Eastern), not UTC.
  Across 212 comparable posts: UTC 19 mismatches, UTC-5 10, UTC-4 zero.
  {date:Olocal/%Y-%m-%d} reproduces it; the trailing separator must be
  omitted or it lands in the strftime format.
- A profile's reels tab returns collab reels owned by OTHER accounts, so
  the directory must be forced with -D. JD2 did the same: chuuo3o and
  official_artms filenames sit inside "0ct0ber19 - reels".
- Stories and highlights need per-item {shortcode}; {post_shortcode} is
  the reel's id and is shared by every item. {date} is per-item, verified
  on a 154-item highlight with distinct times.
- gallery-dl reproduces JD2's caption .txt exactly, including writing
  nothing for an empty caption and omitting the trailing newline.
- The json sidecar needs `include`, not `fields`; `fields` silently does
  nothing in mode:json and leaks audio_user blobs. It yields `type`
  (post/reel) -- Instagram's own flag, which can retire the lone-video
  heuristic once the scanner reads it.

Naming differences between the two tools are cosmetic: EXPORT_RE already
makes the index optional and parseInt normalises zero-padding, so a mixed
archive parses identically. Tests pin that down.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-16 21:04:39 -04:00

159 lines
6.1 KiB
TypeScript

import { describe, expect, it } from 'vitest';
import { parseArchiveFilename, scopedPostId } from './archive-patterns';
describe('parseArchiveFilename — Instagram export format', () => {
it('parses a single-image post', () => {
expect(parseArchiveFilename('2023-04-19_4utumn07 - CrORBIcJJbM.mp4')).toEqual({
postId: 'CrORBIcJJbM',
date: '2023-04-19',
username: '4utumn07',
index: 1,
ext: 'mp4',
isStory: false,
});
});
it('parses a carousel slide index', () => {
const parsed = parseArchiveFilename('2023-04-12_4utumn07 - Cq8LrxSJAJE - 3.jpg');
expect(parsed).toMatchObject({ postId: 'Cq8LrxSJAJE', index: 3, ext: 'jpg' });
});
it('groups a carousel under one post id', () => {
const ids = ['1', '2', '3'].map(
n => parseArchiveFilename(`2023-04-12_user - Cq8LrxSJAJE - ${n}.jpg`)!.postId,
);
expect(new Set(ids).size).toBe(1);
});
it('parses caption sidecar files', () => {
expect(parseArchiveFilename('2023-04-12_4utumn07 - Cq8LrxSJAJE.txt')).toMatchObject({
postId: 'Cq8LrxSJAJE',
ext: 'txt',
});
});
it('flags an explicit story suffix', () => {
expect(parseArchiveFilename('2023-04-12_user - ABC - story.jpg')?.isStory).toBe(true);
});
it('parses the story sidecar layout (date_user - N - shortcode)', () => {
// Files in `story - <user>` carry a per-day ordinal before the shortcode.
const parsed = parseArchiveFilename('2025-10-26_4utumn07 - 2 - DQRuDx9iW5Q.jpg', 'stories');
expect(parsed).toMatchObject({ date: '2025-10-26', username: '4utumn07', ext: 'jpg' });
expect(parsed!.postId).toContain('DQRuDx9iW5Q');
});
it('gives each story item a distinct id', () => {
const a = parseArchiveFilename('2026-08-13_u - 1 - Db-UTJcCUUr.mp4', 'stories')!.postId;
const b = parseArchiveFilename('2026-08-13_u - 2 - Db-oNJ1CWQ4.mp4', 'stories')!.postId;
expect(a).not.toBe(b);
});
});
describe('parseArchiveFilename — Instaloader format', () => {
it('parses a timestamped filename', () => {
expect(parseArchiveFilename('2024-01-01_12-00-00_UTC.jpg')).toMatchObject({
postId: '2024-01-01_12-00-00_UTC',
date: '2024-01-01',
index: 1,
});
});
it('parses the carousel suffix', () => {
expect(parseArchiveFilename('2024-01-01_12-00-00_UTC_2.jpg')?.index).toBe(2);
});
it('flags the story suffix', () => {
expect(parseArchiveFilename('2024-01-01_12-00-00_UTC_story.jpg')?.isStory).toBe(true);
});
});
describe('parseArchiveFilename — story highlights', () => {
it('parses the dateless highlight layout', () => {
expect(parseArchiveFilename('4utumn07 - C5dQPEYpd9W.mp4', 'highlight')).toMatchObject({
postId: 'C5dQPEYpd9W',
username: '4utumn07',
ext: 'mp4',
isStory: false,
});
});
it('dates a highlight from mtime when the filename has none', () => {
const mtime = Date.UTC(2024, 4, 17, 12, 0, 0);
expect(parseArchiveFilename('user - ABC.jpg', 'highlight', mtime)?.date).toBe('2024-05-17');
});
it('leaves the date empty when no mtime is available', () => {
expect(parseArchiveFilename('user - ABC.jpg', 'highlight')?.date).toBe('');
});
it('does not apply the loose highlight pattern outside highlight directories', () => {
// Would otherwise swallow ordinary "a - b.jpg" filenames.
expect(parseArchiveFilename('user - ABC.jpg', 'posts')).toBeNull();
});
});
describe('parseArchiveFilename — non-matching files', () => {
it.each(['4utumn07.jpg', 'profile_pic.jpg', 'README.md', 'no-separator.png'])(
'returns null for %s',
name => expect(parseArchiveFilename(name)).toBeNull(),
);
});
describe('scopedPostId', () => {
it('leaves base-profile ids untouched so permalinks keep working', () => {
expect(scopedPostId('Cq8LrxSJAJE', 'posts')).toBe('Cq8LrxSJAJE');
});
it('namespaces sidecar ids by directory', () => {
expect(scopedPostId('C5dQ', 'highlight', 'story highlights - u - Sunstory'))
.toBe('story highlights - u - Sunstory/C5dQ');
});
it('keeps the same shortcode distinct across sources', () => {
const inPosts = scopedPostId('ABC', 'posts');
const inHighlight = scopedPostId('ABC', 'highlight', 'story highlights - u - H');
expect(inPosts).not.toBe(inHighlight);
});
});
/**
* gallery-dl is replacing JDownloader as the fetcher (docs/gallery-dl.md).
* Its naming differs cosmetically, and these cases pin down that the two
* interoperate so a mixed archive parses identically.
*/
describe('gallery-dl / JDownloader naming interop', () => {
it('treats a single-media post the same with or without an index', () => {
const jd2 = parseArchiveFilename('2023-04-19_4utumn07 - CrORBIcJJbM.mp4')!;
const gdl = parseArchiveFilename('2023-04-19_4utumn07 - CrORBIcJJbM - 1.mp4')!;
expect(jd2.postId).toBe(gdl.postId);
expect(jd2.index).toBe(gdl.index);
expect(jd2.index).toBe(1);
});
it('normalises zero-padded carousel indices', () => {
// JD2 pads to the width of the media count (10+ items -> "01"), and
// gallery-dl's count can be one higher, so the same post may be padded
// by one tool and not the other.
expect(parseArchiveFilename('2024-04-17_4utumn07 - C53YPQzp7Wj - 09.jpg')!.index).toBe(9);
expect(parseArchiveFilename('2024-04-17_4utumn07 - C53YPQzp7Wj - 9.jpg')!.index).toBe(9);
expect(parseArchiveFilename('2023-11-03_4utumn07 - CzM8Uf6B6H_ - 01.jpg')!.index).toBe(1);
});
it('reads a gallery-dl story name, which carries a per-item shortcode', () => {
const p = parseArchiveFilename('2026-08-16_official_band - DcF9OyhBJ1H.jpg', 'stories')!;
expect(p.postId).toBe('DcF9OyhBJ1H');
expect(p.date).toBe('2026-08-16');
});
it('gives a dated highlight a real date instead of the mtime fallback', () => {
const mtime = Date.parse('2026-08-17T00:00:00Z');
const undated = parseArchiveFilename('4utumn07 - C-IImhvpFuk.jpg', 'highlight', mtime)!;
const dated = parseArchiveFilename('2024-08-04_4utumn07 - C-IImhvpFuk.jpg', 'highlight', mtime)!;
// Same item either way, so re-fetching cannot split it into two posts.
expect(dated.postId).toBe(undated.postId);
expect(undated.date).toBe('2026-08-17');
expect(dated.date).toBe('2024-08-04');
});
});