fix: security, performance and correctness pass; add sidecar archive support
Security - Fix path traversal in GET /api/archives/:name/files. Express decodes route params after segment matching, so `..%2f..%2fetc` escaped ARCHIVES_DIR and returned a recursive listing of arbitrary directories. - Add CSP and baseline security headers; disable x-powered-by. - Stop baking GEMINI_API_KEY into the client bundle (the SDK was unused). - Run the container as `node` instead of root. Performance - Add a directory-mtime-keyed archive index, warmed in the background and persisted. Listing 110k files went from ~52s to ~0.1s; the largest archive (24k files) serves in ~0.3s. Per-file stat over CIFS costs ~1.4ms and does not parallelise, so it is now done once rather than per request. - Build media URLs from the File directly instead of `new Blob([await file.arrayBuffer()])`, which read every media file fully into memory (a 20GB archive tried to become 20GB of resident blobs). - Track and revoke object URLs; previously none were ever revoked. - Give `requestThumbnail` a stable identity so a completed thumbnail stops re-running the effect in every mounted thumbnail. - Namespace IndexedDB keys so listing archives no longer deserializes every cached thumbnail blob, and thumbnails no longer collide across archives. - Serve real file sizes: RemoteArchiveFile was constructed with size 0, which silently disabled high-res thumbnailing for every server archive. Correctness - Local archives cached media as blob: URLs, which die with the document, so a cached local archive restored as an archive of broken images. Media now carries a stable path and is rehydrated from a persisted directory handle (File System Access API), falling back to re-prompting for the folder. - Fix permalinks: the URL-writing effect erased ?a= on mount before the archive list arrived to consume it, so deep links never resolved. - Make cache invalidation detect nested changes via a directory signature. - Add an error boundary and tolerate unparseable dates, which previously threw a RangeError and blanked the app. - Default video to muted so autoplay is not blocked by Safari/Firefox. Features - Fold sidecar directories into their base profile: `<user> - reels`, `story - <user>` and `story highlights - <user> - <title>` now appear as reels, the story ring and Instagram-style highlight circles rather than as separate archives. Housekeeping - Add @types/react; React was previously type-checked against its JavaScript source, so `npm run lint` gave almost no type safety on components. - Vendor fonts and PWA icons locally; the app made third-party CDN requests despite advertising offline support and local-only processing. - Drop unused better-sqlite3 (a native module that broke `npm install`). - Add vitest with 36 tests over the filename and directory-naming rules. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_011uBWhwV3wFQ5MBCcMHHem7
This commit is contained in:
co-authored by
Claude Opus 5
parent
0ba7a0d9ad
commit
1d86fa3583
@@ -0,0 +1,170 @@
|
||||
import * as idb from 'idb-keyval';
|
||||
import { CacheData, Post } from '../types';
|
||||
import { DirectoryHandle, ensureReadPermission, filesFromDirectory } from './directory-handle';
|
||||
import { LocalArchiveFile } from './archive-files';
|
||||
|
||||
/**
|
||||
* Persistent archive cache.
|
||||
*
|
||||
* Keys are namespaced so that listing archives does not require deserializing
|
||||
* every thumbnail blob in the store: `archive:` entries are metadata, `thumb:`
|
||||
* entries are image blobs, `handle:` entries are directory handles.
|
||||
*/
|
||||
const ARCHIVE_PREFIX = 'archive:';
|
||||
const THUMB_PREFIX = 'thumb:';
|
||||
const HANDLE_PREFIX = 'handle:';
|
||||
|
||||
export const archiveKey = (name: string) => `${ARCHIVE_PREFIX}${name}`;
|
||||
export const handleKey = (name: string) => `${HANDLE_PREFIX}${name}`;
|
||||
/** Thumbnails are scoped per archive; post IDs alone collide across archives. */
|
||||
export const thumbKey = (archive: string, postId: string) => `${THUMB_PREFIX}${archive}:${postId}`;
|
||||
|
||||
export const getCachedArchive = (name: string): Promise<CacheData | undefined> =>
|
||||
idb.get(archiveKey(name));
|
||||
|
||||
export const setCachedArchive = (data: CacheData) => idb.set(archiveKey(data.name), data);
|
||||
|
||||
/** Names of all cached archives, without loading their contents. */
|
||||
export const listCachedArchiveNames = async (): Promise<string[]> =>
|
||||
(await idb.keys())
|
||||
.map(String)
|
||||
.filter(k => k.startsWith(ARCHIVE_PREFIX))
|
||||
.map(k => k.slice(ARCHIVE_PREFIX.length));
|
||||
|
||||
export const listCachedArchives = async (): Promise<CacheData[]> => {
|
||||
const names = await listCachedArchiveNames();
|
||||
const entries = await Promise.all(names.map(getCachedArchive));
|
||||
return entries.filter((e): e is CacheData => Boolean(e));
|
||||
};
|
||||
|
||||
/** Remove an archive along with its handle and every thumbnail it owns. */
|
||||
export const deleteCachedArchive = async (name: string) => {
|
||||
const thumbPrefix = `${THUMB_PREFIX}${name}:`;
|
||||
const stale = (await idb.keys()).map(String).filter(k => k.startsWith(thumbPrefix));
|
||||
await idb.delMany([archiveKey(name), handleKey(name), ...stale]);
|
||||
};
|
||||
|
||||
export const saveDirectoryHandle = (name: string, handle: DirectoryHandle) =>
|
||||
idb.set(handleKey(name), handle);
|
||||
|
||||
export const getDirectoryHandle = (name: string): Promise<DirectoryHandle | undefined> =>
|
||||
idb.get(handleKey(name));
|
||||
|
||||
/**
|
||||
* One-time migration from the flat key layout (archive name as a bare key,
|
||||
* `thumb_<postId>` for thumbnails).
|
||||
*
|
||||
* Old local entries are dropped rather than migrated: their media URLs are
|
||||
* dead blob: URLs, so restoring them would render an archive of broken images.
|
||||
*/
|
||||
export const migrateLegacyCache = async () => {
|
||||
const keys = (await idb.keys()).map(String);
|
||||
const legacyThumbs = keys.filter(k => k.startsWith('thumb_'));
|
||||
const legacyArchives = keys.filter(
|
||||
k => !k.startsWith(ARCHIVE_PREFIX) && !k.startsWith(THUMB_PREFIX) &&
|
||||
!k.startsWith(HANDLE_PREFIX) && !k.startsWith('thumb_')
|
||||
);
|
||||
if (!legacyThumbs.length && !legacyArchives.length) return;
|
||||
|
||||
const drop: string[] = [...legacyThumbs];
|
||||
for (const key of legacyArchives) {
|
||||
const data = await idb.get(key);
|
||||
drop.push(key);
|
||||
if (data && typeof data === 'object' && 'posts' in data && !(data as CacheData).isLocal) {
|
||||
// Server archives keep working: their URLs are plain HTTP paths.
|
||||
await setCachedArchive({ ...(data as CacheData), name: key });
|
||||
}
|
||||
}
|
||||
await idb.delMany(drop);
|
||||
console.log(`[Cache] Migrated legacy cache: dropped ${drop.length} stale keys.`);
|
||||
};
|
||||
|
||||
/**
|
||||
* Rebuild a server archive's media URLs, which are stable HTTP paths.
|
||||
*
|
||||
* `path` is relative to the archives root and already carries the source
|
||||
* directory (which may be a sidecar such as `story - user`), so it is not
|
||||
* prefixed with the archive name. Entries cached before `path` existed fall
|
||||
* back to their stored URL.
|
||||
*/
|
||||
const rehydrateRemote = (posts: Post[]): Post[] =>
|
||||
posts.map(post => {
|
||||
const media = post.media.map(m => ({
|
||||
...m,
|
||||
url: m.path ? `/archives/${encodeURI(m.path)}` : m.url,
|
||||
}));
|
||||
return { ...post, media, thumbnail: media[0]?.url ?? post.thumbnail };
|
||||
});
|
||||
|
||||
/**
|
||||
* Rebuild a local archive's media URLs from a live directory handle, minting
|
||||
* fresh blob: URLs for the paths recorded at scan time.
|
||||
*
|
||||
* Returns null when the folder is no longer reachable (permission declined, or
|
||||
* the handle no longer resolves), signalling the caller to re-prompt.
|
||||
*/
|
||||
const rehydrateLocal = async (
|
||||
posts: Post[],
|
||||
handle: DirectoryHandle,
|
||||
onUrl: (url: string) => void,
|
||||
): Promise<Post[] | null> => {
|
||||
if (!(await ensureReadPermission(handle))) return null;
|
||||
|
||||
let files: LocalArchiveFile[];
|
||||
try {
|
||||
files = await filesFromDirectory(handle);
|
||||
} catch (err) {
|
||||
console.warn('[Cache] Directory handle no longer readable:', err);
|
||||
return null;
|
||||
}
|
||||
|
||||
const byPath = new Map(files.map(f => [f.webkitRelativePath, f]));
|
||||
|
||||
return posts.map(post => {
|
||||
const media = post.media.map(m => {
|
||||
const file = byPath.get(m.path);
|
||||
if (!file) return { ...m, url: '' };
|
||||
const url = file.createObjectUrl(m.type === 'video' ? 'video/mp4' : 'image/jpeg');
|
||||
onUrl(url);
|
||||
return { ...m, url };
|
||||
});
|
||||
return { ...post, media, thumbnail: media[0]?.url ?? '' };
|
||||
});
|
||||
};
|
||||
|
||||
export interface RestoredArchive {
|
||||
posts: Post[];
|
||||
stories: Post[];
|
||||
highlights: Post[];
|
||||
profileMetadata: CacheData['profileMetadata'];
|
||||
}
|
||||
|
||||
/**
|
||||
* Turn a cache entry back into displayable state.
|
||||
*
|
||||
* `onUrl` receives every blob: URL minted so the caller can revoke them later.
|
||||
* Returns null if a local archive's folder can no longer be reached.
|
||||
*/
|
||||
export const restoreArchive = async (
|
||||
data: CacheData,
|
||||
onUrl: (url: string) => void,
|
||||
): Promise<RestoredArchive | null> => {
|
||||
if (!data.isLocal) {
|
||||
return {
|
||||
posts: rehydrateRemote(data.posts),
|
||||
stories: rehydrateRemote(data.stories),
|
||||
highlights: rehydrateRemote(data.highlights ?? []),
|
||||
profileMetadata: data.profileMetadata,
|
||||
};
|
||||
}
|
||||
|
||||
const handle = await getDirectoryHandle(data.name);
|
||||
if (!handle) return null;
|
||||
|
||||
const posts = await rehydrateLocal(data.posts, handle, onUrl);
|
||||
if (!posts) return null;
|
||||
const stories = (await rehydrateLocal(data.stories, handle, onUrl)) ?? [];
|
||||
const highlights = (await rehydrateLocal(data.highlights ?? [], handle, onUrl)) ?? [];
|
||||
|
||||
return { posts, stories, highlights, profileMetadata: data.profileMetadata };
|
||||
};
|
||||
@@ -1,21 +1,51 @@
|
||||
import { ArchiveFile } from '../types';
|
||||
import { ArchiveFile, ArchiveSource } from '../types';
|
||||
|
||||
export class LocalArchiveFile implements ArchiveFile {
|
||||
constructor(private file: File) {}
|
||||
/** Blob URLs minted here are revocable and must be released when done. */
|
||||
readonly revocable = true;
|
||||
|
||||
/**
|
||||
* @param explicitPath Set when the file came from the File System Access API,
|
||||
* whose File objects carry an empty webkitRelativePath.
|
||||
*/
|
||||
constructor(private file: File, private explicitPath?: string) {}
|
||||
get name() { return this.file.name; }
|
||||
get webkitRelativePath() { return this.file.webkitRelativePath; }
|
||||
get webkitRelativePath() { return this.explicitPath ?? this.file.webkitRelativePath; }
|
||||
get size() { return this.file.size; }
|
||||
text() { return this.file.text(); }
|
||||
arrayBuffer() { return this.file.arrayBuffer(); }
|
||||
stream() { return this.file.stream(); }
|
||||
|
||||
/**
|
||||
* A blob: URL backed directly by the on-disk File.
|
||||
*
|
||||
* Deliberately does NOT go through arrayBuffer() — a File is already a Blob,
|
||||
* so this hands the browser a disk-backed handle instead of pulling the whole
|
||||
* file into memory. Doing otherwise means a 20GB archive tries to become 20GB
|
||||
* of resident blobs.
|
||||
*
|
||||
* When the picker gave us no MIME type, slice() re-tags the blob with a hint.
|
||||
* slice() is a zero-copy view, so this stays memory-free either way.
|
||||
*/
|
||||
createObjectUrl(mimeHint?: string) {
|
||||
const source = this.file.type || !mimeHint
|
||||
? this.file
|
||||
: this.file.slice(0, this.file.size, mimeHint);
|
||||
return URL.createObjectURL(source);
|
||||
}
|
||||
}
|
||||
|
||||
export class RemoteArchiveFile implements ArchiveFile {
|
||||
/** Served over HTTP; there is no object URL to release. */
|
||||
readonly revocable = false;
|
||||
|
||||
constructor(
|
||||
public name: string,
|
||||
public webkitRelativePath: string,
|
||||
public size: number,
|
||||
public url: string
|
||||
public url: string,
|
||||
public source?: ArchiveSource,
|
||||
public mtime?: number
|
||||
) {}
|
||||
async text() {
|
||||
const res = await fetch(this.url);
|
||||
@@ -33,4 +63,8 @@ export class RemoteArchiveFile implements ArchiveFile {
|
||||
});
|
||||
return transform.readable;
|
||||
}
|
||||
|
||||
createObjectUrl() {
|
||||
return this.url;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,89 @@
|
||||
import { describe, expect, it } from 'vitest';
|
||||
import { classifyDirectory, groupArchiveDirectories } from './archive-grouping';
|
||||
|
||||
describe('classifyDirectory', () => {
|
||||
it('treats a bare profile directory as the base', () => {
|
||||
expect(classifyDirectory('4utumn07')).toEqual({
|
||||
owner: '4utumn07',
|
||||
source: { kind: 'posts', dir: '4utumn07' },
|
||||
});
|
||||
});
|
||||
|
||||
it('recognises a reels sidecar', () => {
|
||||
expect(classifyDirectory('4utumn07 - reels')).toEqual({
|
||||
owner: '4utumn07',
|
||||
source: { kind: 'reels', dir: '4utumn07 - reels' },
|
||||
});
|
||||
});
|
||||
|
||||
it('recognises a stories sidecar', () => {
|
||||
expect(classifyDirectory('story - dawn_petal')).toEqual({
|
||||
owner: 'dawn_petal',
|
||||
source: { kind: 'stories', dir: 'story - dawn_petal' },
|
||||
});
|
||||
});
|
||||
|
||||
it('splits highlight owner from title', () => {
|
||||
const { owner, source } = classifyDirectory('story highlights - 4utumn07 - Sunstory');
|
||||
expect(owner).toBe('4utumn07');
|
||||
expect(source.kind).toBe('highlight');
|
||||
expect(source.title).toBe('Sunstory');
|
||||
});
|
||||
|
||||
it.each([
|
||||
['story highlights - theoldlyricmuseinsta - 💙1999-2005 era', 'theoldlyricmuseinsta', '💙1999-2005 era'],
|
||||
['story highlights - member_theworld - [Bracket]', 'member_theworld', '[Bracket]'],
|
||||
['story highlights - official_band - Tour Schedule', 'official_band', 'Tour Schedule'],
|
||||
['story highlights - 4utumn07 - Sketching⠀', '4utumn07', 'Sketching⠀'],
|
||||
['story highlights - official_band - A.B.C', 'official_band', 'A.B.C'],
|
||||
])('handles real-world title %s', (dir, owner, title) => {
|
||||
const result = classifyDirectory(dir);
|
||||
expect(result.owner).toBe(owner);
|
||||
expect(result.source.title).toBe(title);
|
||||
});
|
||||
|
||||
it('keeps titles containing " - " intact', () => {
|
||||
// The username is matched as a non-space run, so only the first separator
|
||||
// splits owner from title.
|
||||
const { owner, source } = classifyDirectory('story highlights - user - a - b');
|
||||
expect(owner).toBe('user');
|
||||
expect(source.title).toBe('a - b');
|
||||
});
|
||||
|
||||
it('does not mistake a profile with spaces for a sidecar', () => {
|
||||
expect(classifyDirectory('Heejin_Bubble heejinmedia').source.kind).toBe('posts');
|
||||
});
|
||||
});
|
||||
|
||||
describe('groupArchiveDirectories', () => {
|
||||
const dirs = [
|
||||
'4utumn07',
|
||||
'4utumn07 - reels',
|
||||
'story - 4utumn07',
|
||||
'story highlights - 4utumn07 - Sunstory',
|
||||
'story highlights - 4utumn07 - Sketching⠀',
|
||||
'kestrelsings',
|
||||
];
|
||||
|
||||
it('folds sidecars into their base profile', () => {
|
||||
const groups = groupArchiveDirectories(dirs);
|
||||
expect([...groups.keys()].sort()).toEqual(['4utumn07', 'kestrelsings']);
|
||||
expect(groups.get('4utumn07')).toHaveLength(5);
|
||||
expect(groups.get('kestrelsings')).toHaveLength(1);
|
||||
});
|
||||
|
||||
it('orders sources posts, reels, stories, then highlights by title', () => {
|
||||
const sources = groupArchiveDirectories(dirs).get('4utumn07')!;
|
||||
expect(sources.map(s => s.kind)).toEqual(['posts', 'reels', 'stories', 'highlight', 'highlight']);
|
||||
expect(sources.slice(3).map(s => s.title)).toEqual(['Sketching⠀', 'Sunstory']);
|
||||
});
|
||||
|
||||
it('still groups a sidecar whose base profile is missing', () => {
|
||||
const groups = groupArchiveDirectories(['story - orphan']);
|
||||
expect(groups.get('orphan')).toEqual([{ kind: 'stories', dir: 'story - orphan' }]);
|
||||
});
|
||||
|
||||
it('is stable for an empty archive root', () => {
|
||||
expect(groupArchiveDirectories([]).size).toBe(0);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,71 @@
|
||||
/**
|
||||
* Archive directory naming rules.
|
||||
*
|
||||
* Shared by the server (to fold sidecar directories into one profile) and the
|
||||
* test suite. Kept free of Node built-ins so it can be imported from either.
|
||||
*/
|
||||
|
||||
export type SourceKind = 'posts' | 'reels' | 'stories' | 'highlight';
|
||||
|
||||
export interface ArchiveSource {
|
||||
kind: SourceKind;
|
||||
/** Directory name relative to the archives root. */
|
||||
dir: string;
|
||||
/** Highlight title, for kind === 'highlight'. */
|
||||
title?: string;
|
||||
}
|
||||
|
||||
/**
|
||||
* Sidecar directories sit next to the profile directory they belong to:
|
||||
*
|
||||
* 4utumn07 -> posts (base)
|
||||
* 4utumn07 - reels -> reels
|
||||
* story - 4utumn07 -> stories
|
||||
* story highlights - 4utumn07 - Sunstory -> highlight "Sunstory"
|
||||
*
|
||||
* Instagram usernames cannot contain spaces, so matching the username as a
|
||||
* run of non-space characters reliably separates it from a highlight title
|
||||
* (titles may themselves contain spaces, dashes and emoji).
|
||||
*/
|
||||
export const classifyDirectory = (dirName: string): { owner: string; source: ArchiveSource } => {
|
||||
const highlight = /^story highlights - ([^ ]+) - (.+)$/.exec(dirName);
|
||||
if (highlight) {
|
||||
return { owner: highlight[1], source: { kind: 'highlight', dir: dirName, title: highlight[2] } };
|
||||
}
|
||||
|
||||
const stories = /^story - ([^ ]+)$/.exec(dirName);
|
||||
if (stories) {
|
||||
return { owner: stories[1], source: { kind: 'stories', dir: dirName } };
|
||||
}
|
||||
|
||||
const reels = /^([^ ]+) - reels$/.exec(dirName);
|
||||
if (reels) {
|
||||
return { owner: reels[1], source: { kind: 'reels', dir: dirName } };
|
||||
}
|
||||
|
||||
return { owner: dirName, source: { kind: 'posts', dir: dirName } };
|
||||
};
|
||||
|
||||
const RANK: Record<SourceKind, number> = { posts: 0, reels: 1, stories: 2, highlight: 3 };
|
||||
|
||||
/**
|
||||
* Group the archive root's directories by profile.
|
||||
*
|
||||
* A sidecar whose owner has no base directory still forms a group of its own,
|
||||
* so nothing becomes invisible just because the base profile is missing.
|
||||
*/
|
||||
export const groupArchiveDirectories = (dirNames: string[]): Map<string, ArchiveSource[]> => {
|
||||
const groups = new Map<string, ArchiveSource[]>();
|
||||
|
||||
for (const dirName of dirNames) {
|
||||
const { owner, source } = classifyDirectory(dirName);
|
||||
if (!groups.has(owner)) groups.set(owner, []);
|
||||
groups.get(owner)!.push(source);
|
||||
}
|
||||
|
||||
for (const sources of groups.values()) {
|
||||
sources.sort((a, b) => RANK[a.kind] - RANK[b.kind] || (a.title ?? '').localeCompare(b.title ?? ''));
|
||||
}
|
||||
|
||||
return groups;
|
||||
};
|
||||
@@ -0,0 +1,232 @@
|
||||
import fs from 'fs';
|
||||
import fsp from 'fs/promises';
|
||||
import path from 'path';
|
||||
import { ArchiveSource, SourceKind, groupArchiveDirectories } from './archive-grouping.js';
|
||||
|
||||
/**
|
||||
* On-disk archive index.
|
||||
*
|
||||
* Archives live on network storage where per-file `stat` costs ~1.4ms and does
|
||||
* not parallelise well, so walking every file on each request is unaffordable:
|
||||
* measured against a real 110k-file archive root, listing took ~52s.
|
||||
*
|
||||
* Directory `stat` is effectively free, so each source directory is indexed
|
||||
* once and re-used until its mtime changes. The index is warmed in the
|
||||
* background at startup and persisted, making steady-state requests instant.
|
||||
*/
|
||||
|
||||
export interface IndexedFile {
|
||||
path: string;
|
||||
size: number;
|
||||
mtime: number;
|
||||
kind: SourceKind;
|
||||
title?: string;
|
||||
}
|
||||
|
||||
interface DirIndex {
|
||||
dir: string;
|
||||
/** Directory mtime the index was built from; the cache key. */
|
||||
mtimeMs: number;
|
||||
files: IndexedFile[];
|
||||
}
|
||||
|
||||
const MEDIA_RE = /\.(jpg|jpeg|png|webp|gif|bmp|tiff|mp4|webm|ogv|mov)$/i;
|
||||
const STAT_CONCURRENCY = 16;
|
||||
|
||||
export class ArchiveIndex {
|
||||
private dirs = new Map<string, DirIndex>();
|
||||
private inFlight = new Map<string, Promise<DirIndex>>();
|
||||
private dirty = false;
|
||||
|
||||
constructor(private archivesDir: string, private cachePath: string) {}
|
||||
|
||||
/** Visible (non-system) directories at the archive root. */
|
||||
private listRootDirs(): string[] {
|
||||
return fs.readdirSync(this.archivesDir, { withFileTypes: true })
|
||||
.filter(e => e.isDirectory() && !/^[.@_]/.test(e.name))
|
||||
.map(e => e.name);
|
||||
}
|
||||
|
||||
groups(): Map<string, ArchiveSource[]> {
|
||||
return groupArchiveDirectories(this.listRootDirs());
|
||||
}
|
||||
|
||||
private dirMtime(dir: string): number {
|
||||
try {
|
||||
return fs.statSync(path.join(this.archivesDir, dir)).mtimeMs;
|
||||
} catch {
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
|
||||
/** Recursively list relative file paths without stat()ing them. */
|
||||
private walk(absDir: string, base = ''): string[] {
|
||||
let out: string[] = [];
|
||||
let entries: fs.Dirent[];
|
||||
try {
|
||||
entries = fs.readdirSync(absDir, { withFileTypes: true });
|
||||
} catch {
|
||||
return out;
|
||||
}
|
||||
for (const entry of entries) {
|
||||
const rel = base ? `${base}/${entry.name}` : entry.name;
|
||||
if (entry.isDirectory()) out = out.concat(this.walk(path.join(absDir, entry.name), rel));
|
||||
else if (entry.isFile()) out.push(rel);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
private async buildDir(source: ArchiveSource): Promise<DirIndex> {
|
||||
const started = Date.now();
|
||||
const mtimeMs = this.dirMtime(source.dir);
|
||||
const absDir = path.join(this.archivesDir, source.dir);
|
||||
const relPaths = this.walk(absDir);
|
||||
|
||||
// Only media needs a size (the client thumbnails anything over 1MiB), and
|
||||
// only highlights need an mtime (their filenames carry no date). Skipping
|
||||
// the rest avoids thousands of pointless round trips.
|
||||
const needsStat = (rel: string) => MEDIA_RE.test(rel) || source.kind === 'highlight';
|
||||
|
||||
const files: IndexedFile[] = relPaths.map(rel => ({
|
||||
path: `${source.dir}/${rel}`,
|
||||
size: 0,
|
||||
mtime: 0,
|
||||
kind: source.kind,
|
||||
...(source.title ? { title: source.title } : {}),
|
||||
}));
|
||||
|
||||
const targets = files.filter((_, i) => needsStat(relPaths[i]));
|
||||
let cursor = 0;
|
||||
const worker = async () => {
|
||||
while (cursor < targets.length) {
|
||||
const file = targets[cursor++];
|
||||
try {
|
||||
const stat = await fsp.stat(path.join(this.archivesDir, file.path));
|
||||
file.size = stat.size;
|
||||
file.mtime = stat.mtimeMs;
|
||||
} catch { /* raced with a delete */ }
|
||||
}
|
||||
};
|
||||
await Promise.all(Array.from({ length: STAT_CONCURRENCY }, worker));
|
||||
|
||||
console.log(
|
||||
`[Index] ${source.dir}: ${files.length} files (${targets.length} statted) in ${((Date.now() - started) / 1000).toFixed(1)}s`
|
||||
);
|
||||
this.dirty = true;
|
||||
return { dir: source.dir, mtimeMs, files };
|
||||
}
|
||||
|
||||
/** Index for one source directory, rebuilding only if its mtime moved. */
|
||||
private async ensureDir(source: ArchiveSource): Promise<DirIndex> {
|
||||
const cached = this.dirs.get(source.dir);
|
||||
const mtimeMs = this.dirMtime(source.dir);
|
||||
if (cached && cached.mtimeMs === mtimeMs) return cached;
|
||||
|
||||
// Collapse concurrent requests for the same directory into one walk.
|
||||
const existing = this.inFlight.get(source.dir);
|
||||
if (existing) return existing;
|
||||
|
||||
const build = this.buildDir(source).then(index => {
|
||||
this.dirs.set(source.dir, index);
|
||||
this.inFlight.delete(source.dir);
|
||||
return index;
|
||||
}).catch(err => {
|
||||
this.inFlight.delete(source.dir);
|
||||
throw err;
|
||||
});
|
||||
this.inFlight.set(source.dir, build);
|
||||
return build;
|
||||
}
|
||||
|
||||
/** All files for one profile, across its base and sidecar directories. */
|
||||
async filesFor(owner: string): Promise<IndexedFile[] | null> {
|
||||
const sources = this.groups().get(owner);
|
||||
if (!sources?.length) return null;
|
||||
const indexes = await Promise.all(sources.map(s => this.ensureDir(s)));
|
||||
return indexes.flatMap(i => i.files);
|
||||
}
|
||||
|
||||
/**
|
||||
* A cheap change signature for a profile, used by the client to decide
|
||||
* whether its cached copy is stale. Built from directory mtimes only, so it
|
||||
* costs one stat per source directory rather than a full walk.
|
||||
*/
|
||||
signatureFor(sources: ArchiveSource[]): string {
|
||||
return sources.map(s => `${s.dir}:${this.dirMtime(s.dir)}`).join('|');
|
||||
}
|
||||
|
||||
/** File count for a profile, if its directories are already indexed. */
|
||||
countFor(sources: ArchiveSource[]): number | null {
|
||||
let total = 0;
|
||||
for (const source of sources) {
|
||||
const cached = this.dirs.get(source.dir);
|
||||
if (!cached) return null;
|
||||
total += cached.files.length;
|
||||
}
|
||||
return total;
|
||||
}
|
||||
|
||||
/**
|
||||
* Best-effort profile picture.
|
||||
*
|
||||
* Probes the conventional filenames first (one stat each) and only falls back
|
||||
* to the indexed listing, so an unindexed archive still gets a thumbnail
|
||||
* without triggering a walk.
|
||||
*/
|
||||
thumbnailFor(owner: string, sources: ArchiveSource[]): string {
|
||||
const base = sources.find(s => s.kind === 'posts') ?? sources[0];
|
||||
if (!base) return '';
|
||||
|
||||
for (const candidate of [`${owner}.jpg`, `${owner}_profile_pic.jpg`, `${owner}.jpeg`, `${owner}.png`]) {
|
||||
if (fs.existsSync(path.join(this.archivesDir, base.dir, candidate))) {
|
||||
return `/archives/${encodeURI(`${base.dir}/${candidate}`)}`;
|
||||
}
|
||||
}
|
||||
|
||||
const cached = this.dirs.get(base.dir);
|
||||
if (cached) {
|
||||
const pick = cached.files.find(f => /_profile_pic\.jpg$/i.test(f.path))
|
||||
?? cached.files.find(f => /\.(jpg|jpeg|png|webp)$/i.test(f.path));
|
||||
if (pick) return `/archives/${encodeURI(pick.path)}`;
|
||||
}
|
||||
return '';
|
||||
}
|
||||
|
||||
/** Walk every directory once, in the background, so first opens are fast. */
|
||||
async warm(): Promise<void> {
|
||||
const started = Date.now();
|
||||
const sources = [...this.groups().values()].flat();
|
||||
console.log(`[Index] Warming ${sources.length} source directories...`);
|
||||
for (const source of sources) {
|
||||
try {
|
||||
await this.ensureDir(source);
|
||||
} catch (err) {
|
||||
console.error(`[Index] Failed to index ${source.dir}:`, err);
|
||||
}
|
||||
}
|
||||
await this.save();
|
||||
console.log(`[Index] Warm complete in ${((Date.now() - started) / 1000).toFixed(1)}s`);
|
||||
}
|
||||
|
||||
async load(): Promise<void> {
|
||||
try {
|
||||
const raw = await fsp.readFile(this.cachePath, 'utf8');
|
||||
const parsed: DirIndex[] = JSON.parse(raw);
|
||||
for (const entry of parsed) this.dirs.set(entry.dir, entry);
|
||||
console.log(`[Index] Loaded ${this.dirs.size} directories from ${this.cachePath}`);
|
||||
} catch {
|
||||
console.log('[Index] No usable index cache; will build from scratch.');
|
||||
}
|
||||
}
|
||||
|
||||
async save(): Promise<void> {
|
||||
if (!this.dirty) return;
|
||||
try {
|
||||
await fsp.writeFile(this.cachePath, JSON.stringify([...this.dirs.values()]), 'utf8');
|
||||
this.dirty = false;
|
||||
console.log(`[Index] Persisted ${this.dirs.size} directories to ${this.cachePath}`);
|
||||
} catch (err) {
|
||||
console.warn('[Index] Could not persist index (continuing in memory):', err);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,118 @@
|
||||
import { describe, expect, it } from 'vitest';
|
||||
import { parseArchiveFilename, scopedPostId } from './archive-patterns';
|
||||
|
||||
describe('parseArchiveFilename — Instagram export format', () => {
|
||||
it('parses a single-image post', () => {
|
||||
expect(parseArchiveFilename('2023-04-19_4utumn07 - CrORBIcJJbM.mp4')).toEqual({
|
||||
postId: 'CrORBIcJJbM',
|
||||
date: '2023-04-19',
|
||||
username: '4utumn07',
|
||||
index: 1,
|
||||
ext: 'mp4',
|
||||
isStory: false,
|
||||
});
|
||||
});
|
||||
|
||||
it('parses a carousel slide index', () => {
|
||||
const parsed = parseArchiveFilename('2023-04-12_4utumn07 - Cq8LrxSJAJE - 3.jpg');
|
||||
expect(parsed).toMatchObject({ postId: 'Cq8LrxSJAJE', index: 3, ext: 'jpg' });
|
||||
});
|
||||
|
||||
it('groups a carousel under one post id', () => {
|
||||
const ids = ['1', '2', '3'].map(
|
||||
n => parseArchiveFilename(`2023-04-12_user - Cq8LrxSJAJE - ${n}.jpg`)!.postId,
|
||||
);
|
||||
expect(new Set(ids).size).toBe(1);
|
||||
});
|
||||
|
||||
it('parses caption sidecar files', () => {
|
||||
expect(parseArchiveFilename('2023-04-12_4utumn07 - Cq8LrxSJAJE.txt')).toMatchObject({
|
||||
postId: 'Cq8LrxSJAJE',
|
||||
ext: 'txt',
|
||||
});
|
||||
});
|
||||
|
||||
it('flags an explicit story suffix', () => {
|
||||
expect(parseArchiveFilename('2023-04-12_user - ABC - story.jpg')?.isStory).toBe(true);
|
||||
});
|
||||
|
||||
it('parses the story sidecar layout (date_user - N - shortcode)', () => {
|
||||
// Files in `story - <user>` carry a per-day ordinal before the shortcode.
|
||||
const parsed = parseArchiveFilename('2025-10-26_4utumn07 - 2 - DQRuDx9iW5Q.jpg', 'stories');
|
||||
expect(parsed).toMatchObject({ date: '2025-10-26', username: '4utumn07', ext: 'jpg' });
|
||||
expect(parsed!.postId).toContain('DQRuDx9iW5Q');
|
||||
});
|
||||
|
||||
it('gives each story item a distinct id', () => {
|
||||
const a = parseArchiveFilename('2026-08-13_u - 1 - Db-UTJcCUUr.mp4', 'stories')!.postId;
|
||||
const b = parseArchiveFilename('2026-08-13_u - 2 - Db-oNJ1CWQ4.mp4', 'stories')!.postId;
|
||||
expect(a).not.toBe(b);
|
||||
});
|
||||
});
|
||||
|
||||
describe('parseArchiveFilename — Instaloader format', () => {
|
||||
it('parses a timestamped filename', () => {
|
||||
expect(parseArchiveFilename('2024-01-01_12-00-00_UTC.jpg')).toMatchObject({
|
||||
postId: '2024-01-01_12-00-00_UTC',
|
||||
date: '2024-01-01',
|
||||
index: 1,
|
||||
});
|
||||
});
|
||||
|
||||
it('parses the carousel suffix', () => {
|
||||
expect(parseArchiveFilename('2024-01-01_12-00-00_UTC_2.jpg')?.index).toBe(2);
|
||||
});
|
||||
|
||||
it('flags the story suffix', () => {
|
||||
expect(parseArchiveFilename('2024-01-01_12-00-00_UTC_story.jpg')?.isStory).toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
describe('parseArchiveFilename — story highlights', () => {
|
||||
it('parses the dateless highlight layout', () => {
|
||||
expect(parseArchiveFilename('4utumn07 - C5dQPEYpd9W.mp4', 'highlight')).toMatchObject({
|
||||
postId: 'C5dQPEYpd9W',
|
||||
username: '4utumn07',
|
||||
ext: 'mp4',
|
||||
isStory: false,
|
||||
});
|
||||
});
|
||||
|
||||
it('dates a highlight from mtime when the filename has none', () => {
|
||||
const mtime = Date.UTC(2024, 4, 17, 12, 0, 0);
|
||||
expect(parseArchiveFilename('user - ABC.jpg', 'highlight', mtime)?.date).toBe('2024-05-17');
|
||||
});
|
||||
|
||||
it('leaves the date empty when no mtime is available', () => {
|
||||
expect(parseArchiveFilename('user - ABC.jpg', 'highlight')?.date).toBe('');
|
||||
});
|
||||
|
||||
it('does not apply the loose highlight pattern outside highlight directories', () => {
|
||||
// Would otherwise swallow ordinary "a - b.jpg" filenames.
|
||||
expect(parseArchiveFilename('user - ABC.jpg', 'posts')).toBeNull();
|
||||
});
|
||||
});
|
||||
|
||||
describe('parseArchiveFilename — non-matching files', () => {
|
||||
it.each(['4utumn07.jpg', 'profile_pic.jpg', 'README.md', 'no-separator.png'])(
|
||||
'returns null for %s',
|
||||
name => expect(parseArchiveFilename(name)).toBeNull(),
|
||||
);
|
||||
});
|
||||
|
||||
describe('scopedPostId', () => {
|
||||
it('leaves base-profile ids untouched so permalinks keep working', () => {
|
||||
expect(scopedPostId('Cq8LrxSJAJE', 'posts')).toBe('Cq8LrxSJAJE');
|
||||
});
|
||||
|
||||
it('namespaces sidecar ids by directory', () => {
|
||||
expect(scopedPostId('C5dQ', 'highlight', 'story highlights - u - Sunstory'))
|
||||
.toBe('story highlights - u - Sunstory/C5dQ');
|
||||
});
|
||||
|
||||
it('keeps the same shortcode distinct across sources', () => {
|
||||
const inPosts = scopedPostId('ABC', 'posts');
|
||||
const inHighlight = scopedPostId('ABC', 'highlight', 'story highlights - u - H');
|
||||
expect(inPosts).not.toBe(inHighlight);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,99 @@
|
||||
import { SourceKind } from '../types';
|
||||
|
||||
/**
|
||||
* Filename parsing rules for the archive formats the viewer understands.
|
||||
*
|
||||
* Kept as pure functions so the riskiest part of the scanner — deriving post
|
||||
* identity, date and carousel order from a filename — can be tested directly.
|
||||
*/
|
||||
|
||||
/** Instagram export: `2023-04-12_user - Cq8LrxSJAJE - 2.jpg` */
|
||||
export const EXPORT_RE = /^(\d{4}-\d{2}-\d{2})_(.+?) - (.+?)(?: - (\d+))?(?: - (story))?\.(.+)$/;
|
||||
|
||||
/** Instaloader: `2024-01-01_12-00-00_UTC_2.jpg` */
|
||||
export const INSTALOADER_RE = /^(\d{4}-\d{2}-\d{2}_\d{2}-\d{2}-\d{2}_UTC)(?:_(\d+))?(?:_(story))?\.(.+)$/;
|
||||
|
||||
/**
|
||||
* Story highlight: `user - C5dQPEYpd9W.mp4` — no date prefix.
|
||||
*
|
||||
* Loose enough to match ordinary filenames, so it is only applied to files the
|
||||
* server has already tagged as coming from a highlight directory.
|
||||
*/
|
||||
export const HIGHLIGHT_RE = /^(.+?) - ([A-Za-z0-9_-]+)\.(\w+)$/;
|
||||
|
||||
export interface ParsedFilename {
|
||||
postId: string;
|
||||
/** ISO date (YYYY-MM-DD), or '' when the filename carries none. */
|
||||
date: string;
|
||||
username: string;
|
||||
/** 1-based carousel position. */
|
||||
index: number;
|
||||
ext: string;
|
||||
isStory: boolean;
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse an archive filename into post identity.
|
||||
*
|
||||
* `kind` selects which patterns apply; `mtime` supplies a date for formats that
|
||||
* have none (highlights), so those items still sort and render sensibly.
|
||||
* Returns null when no pattern matches — e.g. a profile picture.
|
||||
*/
|
||||
export const parseArchiveFilename = (
|
||||
fileName: string,
|
||||
kind: SourceKind = 'posts',
|
||||
mtime?: number,
|
||||
): ParsedFilename | null => {
|
||||
const exp = EXPORT_RE.exec(fileName);
|
||||
if (exp) {
|
||||
const [, date, username, postId, indexStr, story, ext] = exp;
|
||||
return {
|
||||
postId,
|
||||
date,
|
||||
username,
|
||||
index: indexStr ? parseInt(indexStr, 10) : 1,
|
||||
ext,
|
||||
isStory: Boolean(story),
|
||||
};
|
||||
}
|
||||
|
||||
const ins = INSTALOADER_RE.exec(fileName);
|
||||
if (ins) {
|
||||
const [, postId, indexStr, story, ext] = ins;
|
||||
return {
|
||||
postId,
|
||||
date: postId.split('_')[0],
|
||||
username: '',
|
||||
index: indexStr ? parseInt(indexStr, 10) : 1,
|
||||
ext,
|
||||
isStory: Boolean(story),
|
||||
};
|
||||
}
|
||||
|
||||
if (kind === 'highlight') {
|
||||
const hl = HIGHLIGHT_RE.exec(fileName);
|
||||
if (hl) {
|
||||
const [, username, shortcode, ext] = hl;
|
||||
return {
|
||||
postId: shortcode,
|
||||
date: mtime ? new Date(mtime).toISOString().split('T')[0] : '',
|
||||
username,
|
||||
index: 1,
|
||||
ext,
|
||||
isStory: false,
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
return null;
|
||||
};
|
||||
|
||||
/**
|
||||
* Namespace a post ID by its source directory.
|
||||
*
|
||||
* Base-profile IDs are left untouched so existing permalinks keep working;
|
||||
* sidecar IDs are prefixed so a shortcode appearing in both the profile and a
|
||||
* highlight stays two distinct posts.
|
||||
*/
|
||||
export const scopedPostId = (postId: string, kind: SourceKind, dir?: string): string =>
|
||||
kind === 'posts' ? postId : `${dir ?? kind}/${postId}`;
|
||||
@@ -0,0 +1,86 @@
|
||||
import { LocalArchiveFile } from './archive-files';
|
||||
|
||||
/**
|
||||
* File System Access API helpers.
|
||||
*
|
||||
* A `blob:` URL dies with the document, so a cached local archive whose media
|
||||
* URLs are blob: URLs is worthless after a reload. A FileSystemDirectoryHandle,
|
||||
* by contrast, is structured-cloneable and survives in IndexedDB — so we can
|
||||
* re-open the same folder on a return visit and mint fresh URLs from it.
|
||||
*
|
||||
* Only Chromium implements showDirectoryPicker today; callers must handle the
|
||||
* unsupported case by falling back to the <input webkitdirectory> flow.
|
||||
*/
|
||||
|
||||
// Minimal typings — TS's lib.dom does not ship these in the configured version.
|
||||
type PermissionState = 'granted' | 'denied' | 'prompt';
|
||||
interface FileSystemHandlePermissionDescriptor { mode?: 'read' | 'readwrite' }
|
||||
export interface DirectoryHandle {
|
||||
name: string;
|
||||
kind: 'directory';
|
||||
values(): AsyncIterableIterator<DirectoryHandle | FileHandle>;
|
||||
queryPermission?(d?: FileSystemHandlePermissionDescriptor): Promise<PermissionState>;
|
||||
requestPermission?(d?: FileSystemHandlePermissionDescriptor): Promise<PermissionState>;
|
||||
}
|
||||
interface FileHandle {
|
||||
name: string;
|
||||
kind: 'file';
|
||||
getFile(): Promise<File>;
|
||||
}
|
||||
|
||||
export const isDirectoryPickerSupported = () =>
|
||||
typeof window !== 'undefined' && 'showDirectoryPicker' in window;
|
||||
|
||||
export const pickDirectory = async (): Promise<DirectoryHandle | null> => {
|
||||
if (!isDirectoryPickerSupported()) return null;
|
||||
try {
|
||||
return await (window as any).showDirectoryPicker({ mode: 'read' });
|
||||
} catch (err) {
|
||||
// AbortError simply means the user dismissed the picker.
|
||||
return null;
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* Confirm we may still read this handle. Returns false when the user declines
|
||||
* or the grant has lapsed, in which case the caller should re-prompt.
|
||||
*
|
||||
* `requestPermission` must be called from a user gesture, so only call this
|
||||
* while handling a click.
|
||||
*/
|
||||
export const ensureReadPermission = async (handle: DirectoryHandle): Promise<boolean> => {
|
||||
try {
|
||||
if (!handle.queryPermission) return true;
|
||||
if ((await handle.queryPermission({ mode: 'read' })) === 'granted') return true;
|
||||
if (!handle.requestPermission) return false;
|
||||
return (await handle.requestPermission({ mode: 'read' })) === 'granted';
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* Recursively collect every file in the directory.
|
||||
*
|
||||
* Paths are prefixed with the root directory's name so they line up with the
|
||||
* `webkitRelativePath` values produced by <input webkitdirectory>, keeping
|
||||
* cached media paths valid regardless of which picker created them.
|
||||
*/
|
||||
export const filesFromDirectory = async (handle: DirectoryHandle): Promise<LocalArchiveFile[]> => {
|
||||
const out: LocalArchiveFile[] = [];
|
||||
|
||||
const walk = async (dir: DirectoryHandle, prefix: string) => {
|
||||
for await (const entry of dir.values()) {
|
||||
const entryPath = `${prefix}/${entry.name}`;
|
||||
if (entry.kind === 'directory') {
|
||||
await walk(entry as DirectoryHandle, entryPath);
|
||||
} else {
|
||||
const file = await (entry as FileHandle).getFile();
|
||||
out.push(new LocalArchiveFile(file, entryPath));
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
await walk(handle, handle.name);
|
||||
return out;
|
||||
};
|
||||
@@ -1,6 +1,26 @@
|
||||
import { clsx, type ClassValue } from 'clsx';
|
||||
import { twMerge } from 'tailwind-merge';
|
||||
import { format, parseISO } from 'date-fns';
|
||||
|
||||
export function cn(...inputs: ClassValue[]) {
|
||||
return twMerge(clsx(inputs));
|
||||
}
|
||||
|
||||
/**
|
||||
* Format an archive date, tolerating junk.
|
||||
*
|
||||
* Dates are derived from filenames and arbitrary archive JSON, and date-fns
|
||||
* `format` throws a RangeError on an invalid date — which would take down the
|
||||
* whole modal for one malformed name. Story highlights in particular carry no
|
||||
* date at all when file mtimes are unavailable.
|
||||
*/
|
||||
export function formatDateSafe(date: string | undefined, pattern: string): string {
|
||||
if (!date) return '';
|
||||
try {
|
||||
const parsed = parseISO(date);
|
||||
if (Number.isNaN(parsed.getTime())) return '';
|
||||
return format(parsed, pattern);
|
||||
} catch {
|
||||
return '';
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user