Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
26d2d3e379 | ||
|
|
61c2b62141 | ||
|
|
882296b1c0 | ||
|
|
4f8b0021c6 | ||
|
|
dcd8f2ef1d | ||
|
|
5d5dea10c8 | ||
|
|
816fa970b5 | ||
|
|
53b1f80e1d | ||
|
|
89bd5346db |
@@ -9,3 +9,5 @@ coverage/
|
||||
!.env.example
|
||||
_sample-archives
|
||||
_gemini-plans
|
||||
__pycache__/
|
||||
*.pyc
|
||||
|
||||
@@ -15,9 +15,6 @@ InstaArchive Viewer is a React 19 + Vite 6 PWA for browsing archived Instagram d
|
||||
- `npm run lint` — type-check only (`tsc --noEmit`)
|
||||
- `npm test` / `npm run test:watch` — vitest
|
||||
- `npx vitest run src/lib/archive-patterns.test.ts` — a single test file
|
||||
- `npm run jd2 -- --archives <dir> --dry-run` — generate JDownloader `.crawljob`
|
||||
files for every profile on disk (see `scripts/jd2-sync.ts` and
|
||||
`docs/jdownloader.md`)
|
||||
|
||||
Local development usually needs both `npm run dev` and `npm run server`. Local-folder mode works without the backend; server-mode archives do not.
|
||||
|
||||
@@ -64,6 +61,18 @@ story highlights - 4utumn07 - Sunstory -> highlight "Sunstory"
|
||||
|
||||
Results are cached to IndexedDB. Media records store a stable `path`; **`url` is not persistable** for local archives because blob URLs die with the document.
|
||||
|
||||
Three different JSON shapes turn up as `.json`, so they are told apart structurally, not by filename (`src/lib/gallery-dl-sidecar.ts`):
|
||||
|
||||
| shape | marker |
|
||||
|---|---|
|
||||
| Instagram export manifest | top-level `media` array |
|
||||
| Instaloader `.json.xz` | GraphQL node under `node` / `__typename` |
|
||||
| gallery-dl sidecar | flat, `post_shortcode` + `type`, none of the above |
|
||||
|
||||
The gallery-dl sidecar is the only source that states what a post *is*: its `type` (`post` / `reel` / `story` / `highlight`) is Instagram's own classification, so `post.isReel` set from it beats every fallback in `post-tabs.ts`. This matters — of the 781 items in `official_band - reels`, the sidecars say only **360 are reels**; the other 421 are ordinary feed videos the clips endpoint returns via `include_feed_video`. Directory-based classification counted all 781.
|
||||
|
||||
**Dates are ranked, not last-write-wins** (`src/lib/post-dates.ts`): sidecar (what Instagram reported) beats filename (what the fetcher wrote) beats mtime (when the file hit disk, and unrelated to when it was posted). Ties keep the incumbent. Several files describe one post and they are scanned in directory order, not in order of trustworthiness, so without the ranking the date was decided by whichever file came first. Only JDownloader highlights fall to mtime at all — `parseArchiveFilename` flags those via `dateFromMtime`.
|
||||
|
||||
### Cache and local-archive persistence (`src/lib/archive-cache.ts`)
|
||||
|
||||
IndexedDB keys are namespaced (`archive:`, `thumb:`, `handle:`) so listing archives does not deserialize every cached thumbnail blob, and thumbnails are scoped per archive to avoid cross-archive collisions.
|
||||
|
||||
Generated
+2
-2
@@ -1,12 +1,12 @@
|
||||
{
|
||||
"name": "instaarchive-viewer",
|
||||
"version": "1.7.1",
|
||||
"version": "1.8.1",
|
||||
"lockfileVersion": 3,
|
||||
"requires": true,
|
||||
"packages": {
|
||||
"": {
|
||||
"name": "instaarchive-viewer",
|
||||
"version": "1.7.1",
|
||||
"version": "1.8.1",
|
||||
"dependencies": {
|
||||
"@tailwindcss/vite": "^4.1.14",
|
||||
"@vitejs/plugin-react": "^5.0.4",
|
||||
|
||||
+3
-4
@@ -1,19 +1,18 @@
|
||||
{
|
||||
"name": "instaarchive-viewer",
|
||||
"private": true,
|
||||
"version": "1.7.1",
|
||||
"version": "1.8.1",
|
||||
"type": "module",
|
||||
"scripts": {
|
||||
"dev": "vite --port=3000 --host=0.0.0.0",
|
||||
"build": "vite build && npm run build:server",
|
||||
"build:server": "tsc server.ts --esModuleInterop --module ESNext --target ES2022 --moduleResolution bundler --outDir dist-server",
|
||||
"build:server": "tsc server.ts --esModuleInterop --module ESNext --target ES2022 --moduleResolution bundler --removeComments --outDir dist-server",
|
||||
"preview": "vite preview",
|
||||
"server": "tsx server.ts",
|
||||
"clean": "rm -rf dist",
|
||||
"lint": "tsc --noEmit",
|
||||
"test": "vitest run",
|
||||
"test:watch": "vitest",
|
||||
"jd2": "tsx scripts/jd2-sync.ts"
|
||||
"test:watch": "vitest"
|
||||
},
|
||||
"dependencies": {
|
||||
"@tailwindcss/vite": "^4.1.14",
|
||||
|
||||
@@ -4,6 +4,8 @@ import { XzReadableStream } from 'xz-decompress';
|
||||
import { ArchiveFile, CacheData, Post, ServerArchive } from '../types';
|
||||
import { setCachedArchive, getDirectoryHandle } from '../lib/archive-cache';
|
||||
import { parseArchiveFilename, scopedPostId, EXPORT_RE, INSTALOADER_RE } from '../lib/archive-patterns';
|
||||
import { isGalleryDlSidecar, sidecarDate, sidecarIsReel } from '../lib/gallery-dl-sidecar';
|
||||
import { DateSource, shouldReplaceDate } from '../lib/post-dates';
|
||||
|
||||
const hasDirectoryHandle = async (name: string) => Boolean(await getDirectoryHandle(name));
|
||||
|
||||
@@ -142,6 +144,17 @@ export const useArchiveScanner = (
|
||||
|
||||
try {
|
||||
const postsMap = new Map<string, Partial<Post>>();
|
||||
/**
|
||||
* Which source supplied each post's date, so a better one can replace it.
|
||||
* Sidecar beats filename beats mtime — see src/lib/post-dates.ts.
|
||||
*/
|
||||
const dateSources = new Map<string, DateSource>();
|
||||
const applyDate = (postId: string, post: Partial<Post>, date: string, source: DateSource) => {
|
||||
const current = post.date ? { date: post.date, source: dateSources.get(postId) ?? 'mtime' } : undefined;
|
||||
if (!shouldReplaceDate(current, { date, source })) return;
|
||||
post.date = date;
|
||||
dateSources.set(postId, source);
|
||||
};
|
||||
const mediaFilesMap = new Map<string, ArchiveFile>();
|
||||
const discoveredProfilePics: { name: string, url: string }[] = [];
|
||||
const allImageFiles: ArchiveFile[] = [];
|
||||
@@ -300,13 +313,26 @@ export const useArchiveScanner = (
|
||||
}
|
||||
else if (isStory) post.isStory = true;
|
||||
|
||||
// Files describing one post are scanned in directory order, not in
|
||||
// order of trustworthiness, so every date goes through the ranking
|
||||
// in post-dates.ts rather than last-write-wins.
|
||||
applyDate(postId, post, date, parsed.dateFromMtime ? 'mtime' : 'filename');
|
||||
|
||||
const lowerExt = ext.toLowerCase();
|
||||
if (lowerExt === 'txt') {
|
||||
try { post.caption = await file.text(); } catch(e) {}
|
||||
} else if (lowerExt === 'json' || lowerName.endsWith('.json.xz')) {
|
||||
try {
|
||||
const data = lowerName.endsWith('.xz') ? await parseXZFile(file) : JSON.parse(await file.text());
|
||||
if (data) {
|
||||
if (isGalleryDlSidecar(data)) {
|
||||
// The only format that states what a post is rather than
|
||||
// leaving it to be inferred from filenames.
|
||||
if (data.description) post.caption = data.description;
|
||||
const reel = sidecarIsReel(data);
|
||||
if (reel !== undefined) post.isReel = reel;
|
||||
if (data.type === 'story') post.isStory = true;
|
||||
applyDate(postId, post, sidecarDate(data), 'sidecar');
|
||||
} else if (data) {
|
||||
const node = data.node || data; const iphone = node.iphone_struct || {};
|
||||
const captionText = node.edge_media_to_caption?.edges?.[0]?.node?.text || node.caption?.text || iphone.caption?.text || '';
|
||||
if (captionText) post.caption = captionText;
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import { describe, expect, it } from 'vitest';
|
||||
import { parseArchiveFilename, scopedPostId } from './archive-patterns';
|
||||
import { canonicalItemId, parseArchiveFilename, scopedPostId } from './archive-patterns';
|
||||
|
||||
describe('parseArchiveFilename — Instagram export format', () => {
|
||||
it('parses a single-image post', () => {
|
||||
@@ -10,6 +10,7 @@ describe('parseArchiveFilename — Instagram export format', () => {
|
||||
index: 1,
|
||||
ext: 'mp4',
|
||||
isStory: false,
|
||||
dateFromMtime: false,
|
||||
});
|
||||
});
|
||||
|
||||
@@ -116,3 +117,111 @@ describe('scopedPostId', () => {
|
||||
expect(inPosts).not.toBe(inHighlight);
|
||||
});
|
||||
});
|
||||
|
||||
/**
|
||||
* gallery-dl is replacing JDownloader as the fetcher (docs/gallery-dl.md).
|
||||
* Its naming differs cosmetically, and these cases pin down that the two
|
||||
* interoperate so a mixed archive parses identically.
|
||||
*/
|
||||
describe('gallery-dl / JDownloader naming interop', () => {
|
||||
it('treats a single-media post the same with or without an index', () => {
|
||||
const jd2 = parseArchiveFilename('2023-04-19_4utumn07 - CrORBIcJJbM.mp4')!;
|
||||
const gdl = parseArchiveFilename('2023-04-19_4utumn07 - CrORBIcJJbM - 1.mp4')!;
|
||||
expect(jd2.postId).toBe(gdl.postId);
|
||||
expect(jd2.index).toBe(gdl.index);
|
||||
expect(jd2.index).toBe(1);
|
||||
});
|
||||
|
||||
it('normalises zero-padded carousel indices', () => {
|
||||
// JD2 pads to the width of the media count (10+ items -> "01"), and
|
||||
// gallery-dl's count can be one higher, so the same post may be padded
|
||||
// by one tool and not the other.
|
||||
expect(parseArchiveFilename('2024-04-17_4utumn07 - C53YPQzp7Wj - 09.jpg')!.index).toBe(9);
|
||||
expect(parseArchiveFilename('2024-04-17_4utumn07 - C53YPQzp7Wj - 9.jpg')!.index).toBe(9);
|
||||
expect(parseArchiveFilename('2023-11-03_4utumn07 - CzM8Uf6B6H_ - 01.jpg')!.index).toBe(1);
|
||||
});
|
||||
|
||||
it('reads a gallery-dl story name, which carries a per-item shortcode', () => {
|
||||
const p = parseArchiveFilename('2026-08-16_official_band - DcF9OyhBJ1H.jpg', 'stories')!;
|
||||
expect(p.postId).toBe('DcF9OyhBJ1H');
|
||||
expect(p.date).toBe('2026-08-16');
|
||||
});
|
||||
|
||||
it('gives a dated highlight a real date instead of the mtime fallback', () => {
|
||||
const mtime = Date.parse('2026-08-17T00:00:00Z');
|
||||
const undated = parseArchiveFilename('4utumn07 - C-IImhvpFuk.jpg', 'highlight', mtime)!;
|
||||
const dated = parseArchiveFilename('2024-08-04_4utumn07 - C-IImhvpFuk.jpg', 'highlight', mtime)!;
|
||||
// Same item either way, so re-fetching cannot split it into two posts.
|
||||
expect(dated.postId).toBe(undated.postId);
|
||||
expect(undated.date).toBe('2026-08-17');
|
||||
expect(dated.date).toBe('2024-08-04');
|
||||
});
|
||||
});
|
||||
|
||||
/**
|
||||
* Highlights are the only files with no date in the name, so they fall back to
|
||||
* mtime — which is when the file was written, not when it was posted. Callers
|
||||
* need to know the difference to let a real date win.
|
||||
*/
|
||||
describe('dateFromMtime', () => {
|
||||
const mtime = Date.parse('2026-08-17T00:00:00Z');
|
||||
|
||||
it('flags an undated highlight name as mtime-dated', () => {
|
||||
const p = parseArchiveFilename('4utumn07 - C-IImhvpFuk.jpg', 'highlight', mtime)!;
|
||||
expect(p.date).toBe('2026-08-17');
|
||||
expect(p.dateFromMtime).toBe(true);
|
||||
});
|
||||
|
||||
it('does not flag a highlight that carries its own date', () => {
|
||||
const p = parseArchiveFilename('2024-08-04_4utumn07 - C-IImhvpFuk.jpg', 'highlight', mtime)!;
|
||||
expect(p.date).toBe('2024-08-04');
|
||||
expect(p.dateFromMtime).toBe(false);
|
||||
});
|
||||
|
||||
it('never flags ordinary post or Instaloader names', () => {
|
||||
expect(parseArchiveFilename('2023-04-19_u - ABC.mp4', 'posts', mtime)!.dateFromMtime).toBe(false);
|
||||
expect(parseArchiveFilename('2024-01-01_12-00-00_UTC.jpg', 'posts', mtime)!.dateFromMtime).toBe(false);
|
||||
});
|
||||
|
||||
it('leaves the date empty rather than guessing when no mtime is given', () => {
|
||||
const p = parseArchiveFilename('4utumn07 - C-IImhvpFuk.jpg', 'highlight')!;
|
||||
expect(p.date).toBe('');
|
||||
expect(p.dateFromMtime).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
/**
|
||||
* JDownloader wrote story-shaped names for highlights during one period, so
|
||||
* the same item exists under two conventions. They must be one post.
|
||||
*/
|
||||
describe('canonicalItemId', () => {
|
||||
it('collapses the two highlight naming conventions onto one id', () => {
|
||||
const dir = 'story highlights - 4utumn07 - Sunstory';
|
||||
const undated = parseArchiveFilename('4utumn07 - C5dQPEYpd9W.mp4', 'highlight', 1)!;
|
||||
const dated = parseArchiveFilename('2024-04-07_4utumn07 - 01 - C5dQPEYpd9W.mp4', 'highlight')!;
|
||||
expect(scopedPostId(dated.postId, 'highlight', dir))
|
||||
.toBe(scopedPostId(undated.postId, 'highlight', dir));
|
||||
});
|
||||
|
||||
it('does the same for stories', () => {
|
||||
const a = parseArchiveFilename('2025-10-26_u - 2 - DQRuDx9iW5Q.jpg', 'stories')!;
|
||||
expect(scopedPostId(a.postId, 'stories', 'story - u')).toBe('story - u/DQRuDx9iW5Q');
|
||||
});
|
||||
|
||||
it('keeps distinct story items distinct', () => {
|
||||
const a = parseArchiveFilename('2026-08-13_u - 1 - Db-UTJcCUUr.mp4', 'stories')!;
|
||||
const b = parseArchiveFilename('2026-08-13_u - 2 - Db-oNJ1CWQ4.mp4', 'stories')!;
|
||||
expect(scopedPostId(a.postId, 'stories', 'story - u'))
|
||||
.not.toBe(scopedPostId(b.postId, 'stories', 'story - u'));
|
||||
});
|
||||
|
||||
it('leaves a shortcode that merely starts with digits alone', () => {
|
||||
expect(canonicalItemId('4utumn07')).toBe('4utumn07');
|
||||
expect(canonicalItemId('C5dQPEYpd9W')).toBe('C5dQPEYpd9W');
|
||||
expect(canonicalItemId('12345')).toBe('12345');
|
||||
});
|
||||
|
||||
it('does not touch posts, whose ids are permalinks', () => {
|
||||
expect(scopedPostId('01 - ABC', 'posts')).toBe('01 - ABC');
|
||||
});
|
||||
});
|
||||
|
||||
@@ -30,6 +30,15 @@ export interface ParsedFilename {
|
||||
index: number;
|
||||
ext: string;
|
||||
isStory: boolean;
|
||||
/**
|
||||
* True when `date` is the file's mtime rather than anything Instagram said.
|
||||
*
|
||||
* Only highlights fetched by JDownloader lack a date in the filename, and
|
||||
* their mtime is just when the file was written. Callers should let any real
|
||||
* date win over this one — the same item is often also present under a
|
||||
* gallery-dl name that does carry the date.
|
||||
*/
|
||||
dateFromMtime: boolean;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -54,6 +63,7 @@ export const parseArchiveFilename = (
|
||||
index: indexStr ? parseInt(indexStr, 10) : 1,
|
||||
ext,
|
||||
isStory: Boolean(story),
|
||||
dateFromMtime: false,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -67,6 +77,7 @@ export const parseArchiveFilename = (
|
||||
index: indexStr ? parseInt(indexStr, 10) : 1,
|
||||
ext,
|
||||
isStory: Boolean(story),
|
||||
dateFromMtime: false,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -81,6 +92,7 @@ export const parseArchiveFilename = (
|
||||
index: 1,
|
||||
ext,
|
||||
isStory: false,
|
||||
dateFromMtime: Boolean(mtime),
|
||||
};
|
||||
}
|
||||
}
|
||||
@@ -88,12 +100,36 @@ export const parseArchiveFilename = (
|
||||
return null;
|
||||
};
|
||||
|
||||
/**
|
||||
* A leading per-day ordinal on a story or highlight id: `01 - C5dQPEYpd9W`.
|
||||
*
|
||||
* JDownloader wrote story-shaped names for highlights during one period of its
|
||||
* life, so the same item exists as both `user - CODE.jpg` and
|
||||
* `date_user - 01 - CODE.jpg`. Those parse to different ids and the viewer
|
||||
* shows the item twice. The ordinal carries no information the shortcode does
|
||||
* not — it is a position within a day's stories, and the shortcode is already
|
||||
* unique — so it is dropped.
|
||||
*/
|
||||
const LEADING_ORDINAL = /^\d+ - (?=[A-Za-z0-9_-]+$)/;
|
||||
|
||||
/** Strip the ordinal so both naming conventions land on the same post. */
|
||||
export const canonicalItemId = (postId: string): string =>
|
||||
postId.replace(LEADING_ORDINAL, '');
|
||||
|
||||
/**
|
||||
* Namespace a post ID by its source directory.
|
||||
*
|
||||
* Base-profile IDs are left untouched so existing permalinks keep working;
|
||||
* sidecar IDs are prefixed so a shortcode appearing in both the profile and a
|
||||
* highlight stays two distinct posts.
|
||||
*
|
||||
* Story and highlight ids are canonicalised first, so an item fetched under
|
||||
* two different naming conventions is one post rather than two.
|
||||
*/
|
||||
export const scopedPostId = (postId: string, kind: SourceKind, dir?: string): string =>
|
||||
kind === 'posts' ? postId : `${dir ?? kind}/${postId}`;
|
||||
export const scopedPostId = (postId: string, kind: SourceKind, dir?: string): string => {
|
||||
if (kind === 'posts') return postId;
|
||||
const id = (kind === 'stories' || kind === 'highlight')
|
||||
? canonicalItemId(postId)
|
||||
: postId;
|
||||
return `${dir ?? kind}/${id}`;
|
||||
};
|
||||
|
||||
@@ -0,0 +1,82 @@
|
||||
import { describe, expect, it } from 'vitest';
|
||||
import {
|
||||
GalleryDlSidecar, isGalleryDlSidecar, sidecarDate, sidecarIsReel, sidecarSource,
|
||||
} from './gallery-dl-sidecar';
|
||||
|
||||
// Trimmed from real files published to the archive on 2026-08-16.
|
||||
const REEL: GalleryDlSidecar = {
|
||||
post_shortcode: 'Db-lNCoib9m', post_id: '3962768346034323302', type: 'reel',
|
||||
date: '2026-08-13 11:00:44', post_date: '2026-08-13 11:00:44',
|
||||
username: 'official_band', fullname: 'Official ARTMS',
|
||||
description: 'Dancing in the spotlight', count: 1, likes: 22914,
|
||||
};
|
||||
const FEED_VIDEO: GalleryDlSidecar = { ...REEL, post_shortcode: 'DbdG9L9jU4m', type: 'post', count: 2 };
|
||||
const HIGHLIGHT: GalleryDlSidecar = {
|
||||
post_shortcode: 'BATVdRZi_3', post_id: '18099435932626935', type: 'highlight',
|
||||
date: '2026-08-08 16:22:09', username: 'official_band', count: 154,
|
||||
};
|
||||
|
||||
describe('isGalleryDlSidecar', () => {
|
||||
it('accepts a real sidecar', () => {
|
||||
expect(isGalleryDlSidecar(REEL)).toBe(true);
|
||||
expect(isGalleryDlSidecar(HIGHLIGHT)).toBe(true);
|
||||
});
|
||||
|
||||
it('rejects an Instaloader GraphQL payload', () => {
|
||||
expect(isGalleryDlSidecar({ node: { __typename: 'GraphVideo', shortcode: 'x' } })).toBe(false);
|
||||
expect(isGalleryDlSidecar({ __typename: 'GraphImage', post_shortcode: 'x', type: 'post' })).toBe(false);
|
||||
});
|
||||
|
||||
it('rejects an Instagram export manifest', () => {
|
||||
expect(isGalleryDlSidecar({ media: [{ uri: 'a.jpg' }] })).toBe(false);
|
||||
expect(isGalleryDlSidecar([{ media: [] }])).toBe(false);
|
||||
});
|
||||
|
||||
it('rejects junk', () => {
|
||||
for (const v of [null, undefined, 0, '', 'string', {}, { post_shortcode: 'x' }]) {
|
||||
expect(isGalleryDlSidecar(v)).toBe(false);
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe('sidecarDate', () => {
|
||||
it('takes the day from the timestamp', () => {
|
||||
expect(sidecarDate(REEL)).toBe('2026-08-13');
|
||||
});
|
||||
|
||||
it('falls back to post_date', () => {
|
||||
expect(sidecarDate({ post_shortcode: 'x', post_date: '2024-01-02 03:04:05' })).toBe('2024-01-02');
|
||||
});
|
||||
|
||||
it('returns empty when there is no usable date', () => {
|
||||
expect(sidecarDate({ post_shortcode: 'x' })).toBe('');
|
||||
expect(sidecarDate({ post_shortcode: 'x', date: 'not a date' })).toBe('');
|
||||
});
|
||||
});
|
||||
|
||||
describe('sidecarIsReel', () => {
|
||||
it('distinguishes a reel from an ordinary feed video', () => {
|
||||
// Both are single mp4s -- the lone-video heuristic cannot tell them apart.
|
||||
expect(sidecarIsReel(REEL)).toBe(true);
|
||||
expect(sidecarIsReel(FEED_VIDEO)).toBe(false);
|
||||
});
|
||||
|
||||
it('declines to answer for stories and highlights', () => {
|
||||
expect(sidecarIsReel(HIGHLIGHT)).toBeUndefined();
|
||||
expect(sidecarIsReel({ post_shortcode: 'x', type: 'story' as const })).toBeUndefined();
|
||||
expect(sidecarIsReel({ post_shortcode: 'x' })).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
describe('sidecarSource', () => {
|
||||
it('maps type onto the archive source kinds', () => {
|
||||
expect(sidecarSource(REEL)).toBe('reels');
|
||||
expect(sidecarSource(FEED_VIDEO)).toBe('posts');
|
||||
expect(sidecarSource(HIGHLIGHT)).toBe('highlight');
|
||||
expect(sidecarSource({ post_shortcode: 'x', type: 'story' as const })).toBe('stories');
|
||||
});
|
||||
|
||||
it('is undefined for an unknown type', () => {
|
||||
expect(sidecarSource({ post_shortcode: 'x' })).toBeUndefined();
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,87 @@
|
||||
import { SourceKind } from '../types';
|
||||
|
||||
/**
|
||||
* gallery-dl `.json` metadata sidecars.
|
||||
*
|
||||
* Written one per post next to the media (see docs/gallery-dl.md). This is the
|
||||
* only source in any archive format that states outright what a post *is* —
|
||||
* `type` is Instagram's own classification, the `product_type: "clips"` signal
|
||||
* carried through the listing response. Everything else the viewer knows about
|
||||
* reels is guesswork from filenames and directory names.
|
||||
*
|
||||
* Deliberately separate from the two older JSON shapes the scanner reads:
|
||||
*
|
||||
* Instagram export `posts_1.json`, an array of entries with `media`
|
||||
* Instaloader `.json.xz`, a GraphQL node under `node`
|
||||
* gallery-dl this — flat, no wrapper
|
||||
*/
|
||||
export interface GalleryDlSidecar {
|
||||
post_shortcode: string;
|
||||
post_id?: string;
|
||||
/** Instagram's own classification of the post. */
|
||||
type?: 'post' | 'reel' | 'story' | 'highlight';
|
||||
/** Local-time "YYYY-MM-DD HH:MM:SS" — gallery-dl is configured to emit local. */
|
||||
date?: string;
|
||||
post_date?: string;
|
||||
username?: string;
|
||||
fullname?: string;
|
||||
description?: string;
|
||||
count?: number;
|
||||
likes?: number;
|
||||
post_url?: string;
|
||||
}
|
||||
|
||||
/**
|
||||
* Recognise a gallery-dl sidecar.
|
||||
*
|
||||
* Checked structurally rather than by filename, because the older formats are
|
||||
* also plain `.json`. `node` and `__typename` are what an Instaloader or
|
||||
* export payload carries, and their absence is what makes this shape
|
||||
* unambiguous.
|
||||
*/
|
||||
export const isGalleryDlSidecar = (data: unknown): data is GalleryDlSidecar => {
|
||||
if (!data || typeof data !== 'object' || Array.isArray(data)) return false;
|
||||
const o = data as Record<string, unknown>;
|
||||
return typeof o.post_shortcode === 'string'
|
||||
&& typeof o.type === 'string'
|
||||
&& o.node === undefined
|
||||
&& o.__typename === undefined
|
||||
&& o.media === undefined;
|
||||
};
|
||||
|
||||
/** The ISO date (YYYY-MM-DD) a sidecar reports, or '' if it carries none. */
|
||||
export const sidecarDate = (s: GalleryDlSidecar): string => {
|
||||
const raw = s.date || s.post_date || '';
|
||||
const day = raw.slice(0, 10);
|
||||
return /^\d{4}-\d{2}-\d{2}$/.test(day) ? day : '';
|
||||
};
|
||||
|
||||
/**
|
||||
* Whether the sidecar says this post is a reel.
|
||||
*
|
||||
* Returns undefined rather than false for stories and highlights: those are
|
||||
* neither reels nor grid posts, and answering "no" would let them be counted
|
||||
* as ordinary posts.
|
||||
*/
|
||||
export const sidecarIsReel = (s: GalleryDlSidecar): boolean | undefined => {
|
||||
if (s.type === 'reel') return true;
|
||||
if (s.type === 'post') return false;
|
||||
return undefined;
|
||||
};
|
||||
|
||||
/**
|
||||
* Which source kind the sidecar implies, for cross-checking the directory.
|
||||
*
|
||||
* A reel shared to the profile grid legitimately appears under `posts`, so a
|
||||
* disagreement is not an error — the directory says where the file was
|
||||
* fetched from, `type` says what Instagram considers it.
|
||||
*/
|
||||
export const sidecarSource = (s: GalleryDlSidecar): SourceKind | undefined => {
|
||||
switch (s.type) {
|
||||
case 'reel': return 'reels';
|
||||
case 'post': return 'posts';
|
||||
case 'story': return 'stories';
|
||||
case 'highlight': return 'highlight';
|
||||
default: return undefined;
|
||||
}
|
||||
};
|
||||
@@ -0,0 +1,56 @@
|
||||
import { describe, expect, it } from 'vitest';
|
||||
import { DatedValue, preferDate, shouldReplaceDate } from './post-dates';
|
||||
|
||||
const sidecar: DatedValue = { date: '2024-04-07', source: 'sidecar' };
|
||||
const filename: DatedValue = { date: '2024-04-08', source: 'filename' };
|
||||
const mtime: DatedValue = { date: '2026-08-17', source: 'mtime' };
|
||||
|
||||
describe('date precedence', () => {
|
||||
it('ranks sidecar above filename above mtime', () => {
|
||||
expect(preferDate(mtime, filename)).toEqual(filename);
|
||||
expect(preferDate(filename, sidecar)).toEqual(sidecar);
|
||||
expect(preferDate(mtime, sidecar)).toEqual(sidecar);
|
||||
});
|
||||
|
||||
it('never lets a weaker source overwrite a stronger one', () => {
|
||||
expect(preferDate(sidecar, filename)).toEqual(sidecar);
|
||||
expect(preferDate(sidecar, mtime)).toEqual(sidecar);
|
||||
expect(preferDate(filename, mtime)).toEqual(filename);
|
||||
});
|
||||
|
||||
it('keeps the incumbent on a tie, so scan order cannot flip the date', () => {
|
||||
const other: DatedValue = { date: '2020-01-01', source: 'filename' };
|
||||
expect(preferDate(filename, other)).toEqual(filename);
|
||||
expect(preferDate(other, filename)).toEqual(other);
|
||||
});
|
||||
|
||||
it('accepts anything when nothing is held yet', () => {
|
||||
expect(preferDate(undefined, mtime)).toEqual(mtime);
|
||||
expect(shouldReplaceDate(undefined, mtime)).toBe(true);
|
||||
});
|
||||
|
||||
it('ignores an empty date regardless of source', () => {
|
||||
const empty: DatedValue = { date: '', source: 'sidecar' };
|
||||
expect(shouldReplaceDate(filename, empty)).toBe(false);
|
||||
expect(preferDate(filename, empty)).toEqual(filename);
|
||||
});
|
||||
|
||||
it('replaces a held-but-empty date', () => {
|
||||
const empty: DatedValue = { date: '', source: 'filename' };
|
||||
expect(preferDate(empty, mtime)).toEqual(mtime);
|
||||
});
|
||||
|
||||
it('is order-independent for the full three-source case', () => {
|
||||
const orders = [
|
||||
[mtime, filename, sidecar],
|
||||
[sidecar, mtime, filename],
|
||||
[filename, sidecar, mtime],
|
||||
[mtime, sidecar, filename],
|
||||
];
|
||||
for (const order of orders) {
|
||||
const won = order.reduce<DatedValue | undefined>(
|
||||
(acc, next) => preferDate(acc, next), undefined);
|
||||
expect(won).toEqual(sidecar);
|
||||
}
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,46 @@
|
||||
/**
|
||||
* Where a post's date came from, and which source wins.
|
||||
*
|
||||
* A post is usually described by several files — media, a caption `.txt`, a
|
||||
* `.json` sidecar, sometimes the same item under two naming conventions — and
|
||||
* they are scanned in directory order, not in order of trustworthiness. Without
|
||||
* an explicit ranking the date is decided by whichever file happened to be
|
||||
* reached first.
|
||||
*
|
||||
* Ranked best to worst:
|
||||
*
|
||||
* sidecar what Instagram reported, straight from a gallery-dl `.json`
|
||||
* filename a date the fetcher wrote into the name; correct, but derived
|
||||
* mtime when the file was written to disk — unrelated to when it was
|
||||
* posted, and only ever a last resort for JDownloader highlights,
|
||||
* whose filenames carry no date at all
|
||||
*/
|
||||
export type DateSource = 'sidecar' | 'filename' | 'mtime';
|
||||
|
||||
const RANK: Record<DateSource, number> = { sidecar: 0, filename: 1, mtime: 2 };
|
||||
|
||||
export interface DatedValue {
|
||||
date: string;
|
||||
source: DateSource;
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether `next` should replace the date currently held.
|
||||
*
|
||||
* Ties keep the incumbent, so scanning stays stable: two files of equal
|
||||
* authority cannot flip a post's date back and forth by scan order.
|
||||
*/
|
||||
export const shouldReplaceDate = (
|
||||
current: DatedValue | undefined,
|
||||
next: DatedValue,
|
||||
): boolean => {
|
||||
if (!next.date) return false;
|
||||
if (!current || !current.date) return true;
|
||||
return RANK[next.source] < RANK[current.source];
|
||||
};
|
||||
|
||||
/** Apply `next` if it outranks `current`, otherwise keep what we have. */
|
||||
export const preferDate = (
|
||||
current: DatedValue | undefined,
|
||||
next: DatedValue,
|
||||
): DatedValue => (shouldReplaceDate(current, next) ? next : (current ?? next));
|
||||
@@ -98,3 +98,44 @@ describe('postsForTab', () => {
|
||||
expect(postsForTab([post('A')], 'saved')).toEqual([]);
|
||||
});
|
||||
});
|
||||
|
||||
/**
|
||||
* Once an archive carries gallery-dl sidecars, the guesswork above is replaced
|
||||
* by Instagram's own classification. These are the cases the heuristic got
|
||||
* wrong (see docs/gallery-dl.md).
|
||||
*/
|
||||
describe('explicit isReel from a sidecar', () => {
|
||||
it('beats the lone-video heuristic for an ordinary feed video', () => {
|
||||
// A single mp4 that Instagram calls a post, not a reel — indistinguishable
|
||||
// by shape alone.
|
||||
const posts = [video('DbdG9L9jU4m', { isReel: false })];
|
||||
expect(postsForTab(posts, 'reels')).toEqual([]);
|
||||
expect(postsForTab(posts, 'posts')).toHaveLength(1);
|
||||
});
|
||||
|
||||
it('recognises a reel that lives in the profile grid', () => {
|
||||
// Shared to feed, so it sits in the base directory with source 'posts'.
|
||||
const posts = [post('A'), video('C8FHM6EJl15', { source: 'posts', isReel: true })];
|
||||
expect(postsForTab(posts, 'reels').map(p => p.id)).toEqual(['C8FHM6EJl15']);
|
||||
expect(postsForTab(posts, 'posts')).toHaveLength(2);
|
||||
});
|
||||
|
||||
it('beats the directory when both are present', () => {
|
||||
const posts = [
|
||||
video('u - reels/A', { source: 'reels', isReel: false }),
|
||||
video('u - reels/B', { source: 'reels' }),
|
||||
];
|
||||
// A is a feed video that the reels tab happened to return; B is unlabelled
|
||||
// and falls back to its directory.
|
||||
expect(postsForTab(posts, 'reels').map(p => p.id)).toEqual(['u - reels/B']);
|
||||
});
|
||||
|
||||
it('falls back per post, so a mixed archive still works', () => {
|
||||
const posts = [
|
||||
video('labelled', { isReel: true }),
|
||||
video('unlabelled'),
|
||||
carousel('C'),
|
||||
];
|
||||
expect(postsForTab(posts, 'reels').map(p => p.id)).toEqual(['labelled', 'unlabelled']);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -41,11 +41,16 @@ export const hasReelSource = (posts: Post[]): boolean => posts.some(p => p.sourc
|
||||
* or an old IGTV upload, all three of which are plain `GraphVideo` nodes
|
||||
* distinguished only by `product_type`.
|
||||
*/
|
||||
export const makeIsReel = (posts: Post[]): ((post: Post) => boolean) => (
|
||||
hasReelSource(posts)
|
||||
export const makeIsReel = (posts: Post[]): ((post: Post) => boolean) => {
|
||||
const guess = hasReelSource(posts)
|
||||
? (post: Post) => post.source === 'reels'
|
||||
: (post: Post) => post.media.length === 1 && post.media[0]?.type === 'video'
|
||||
);
|
||||
: (post: Post) => post.media.length === 1 && post.media[0]?.type === 'video';
|
||||
|
||||
// `isReel` comes from a gallery-dl sidecar and is Instagram's own answer, so
|
||||
// it beats both fallbacks — per post, since an archive is usually a mix of
|
||||
// files fetched before and after sidecars existed.
|
||||
return (post: Post) => post.isReel ?? guess(post);
|
||||
};
|
||||
|
||||
/** Preference order when the same post was fetched into more than one directory. */
|
||||
const SOURCE_RANK: Record<SourceKind, number> = { reels: 0, posts: 1, stories: 2, highlight: 3 };
|
||||
|
||||
@@ -37,6 +37,12 @@ export interface Post {
|
||||
isStory?: boolean;
|
||||
/** Defaults to 'posts' for archives without sidecar directories. */
|
||||
source?: SourceKind;
|
||||
/**
|
||||
* Instagram's own answer to "is this a reel", from a gallery-dl `.json`
|
||||
* sidecar. Undefined when the archive carries no such sidecar, which is when
|
||||
* the viewer has to fall back to guessing — see src/lib/post-tabs.ts.
|
||||
*/
|
||||
isReel?: boolean;
|
||||
/** Highlight this post belongs to, for source === 'highlight'. */
|
||||
highlightTitle?: string;
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user