mirror of
https://github.com/thegeneralist01/archivr
synced 2026-10-09 12:55:00 +02:00
Compare commits
No commits in common. "094f1b045774c288d7679627b99748232124f1be" and "bf5f95397fc83652f023670b14da508b894ef568" have entirely different histories.
094f1b0457
...
bf5f95397f
61 changed files with 2722 additions and 19420 deletions
131
.github/workflows/update-ytdlp.yml
vendored
131
.github/workflows/update-ytdlp.yml
vendored
|
|
@ -13,126 +13,39 @@ jobs:
|
|||
pull-requests: write
|
||||
|
||||
steps:
|
||||
- name: Check out repository
|
||||
uses: actions/checkout@v4
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Install Nix
|
||||
uses: DeterminateSystems/nix-installer-action@main
|
||||
- uses: DeterminateSystems/nix-installer-action@main
|
||||
|
||||
- name: Enable Nix cache
|
||||
uses: DeterminateSystems/magic-nix-cache-action@main
|
||||
- uses: DeterminateSystems/magic-nix-cache-action@main
|
||||
|
||||
- name: Read currently pinned yt-dlp version
|
||||
id: current
|
||||
- name: Get current yt-dlp version
|
||||
id: before
|
||||
run: |
|
||||
set -euo pipefail
|
||||
current=$(sed -n '/ytDlp = pkgs.stdenv.mkDerivation/,/^ };$/{ s/^ *version = "\([^"]*\)";/\1/p; }' flake.nix | head -1)
|
||||
if [ -z "$current" ]; then
|
||||
echo "::error::Could not read the pinned yt-dlp version from flake.nix. Did the ytDlp derivation move or get renamed?"
|
||||
exit 1
|
||||
fi
|
||||
echo "Currently pinned yt-dlp: $current"
|
||||
echo "version=${current}" >> "$GITHUB_OUTPUT"
|
||||
rev=$(jq -r '.nodes.nixpkgs.locked.rev' flake.lock)
|
||||
version=$(nix eval --raw "github:nixos/nixpkgs/${rev}#yt-dlp.version")
|
||||
echo "version=${version}" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Query latest yt-dlp release
|
||||
id: latest
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
- name: Update nixpkgs
|
||||
run: nix flake update nixpkgs
|
||||
|
||||
- name: Get new yt-dlp version
|
||||
id: after
|
||||
run: |
|
||||
set -euo pipefail
|
||||
latest=$(curl -sSL \
|
||||
-H "Authorization: Bearer $GITHUB_TOKEN" \
|
||||
-H "Accept: application/vnd.github+json" \
|
||||
https://api.github.com/repos/yt-dlp/yt-dlp/releases/latest | jq -r .tag_name)
|
||||
if [ -z "$latest" ] || [ "$latest" = "null" ]; then
|
||||
echo "::error::Could not determine the latest yt-dlp release tag from the GitHub API."
|
||||
exit 1
|
||||
fi
|
||||
echo "Latest yt-dlp release: $latest"
|
||||
echo "version=${latest}" >> "$GITHUB_OUTPUT"
|
||||
rev=$(jq -r '.nodes.nixpkgs.locked.rev' flake.lock)
|
||||
version=$(nix eval --raw "github:nixos/nixpkgs/${rev}#yt-dlp.version")
|
||||
echo "version=${version}" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Decide whether an update is needed
|
||||
id: check
|
||||
env:
|
||||
CURRENT: ${{ steps.current.outputs.version }}
|
||||
LATEST: ${{ steps.latest.outputs.version }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ "$CURRENT" = "$LATEST" ]; then
|
||||
echo "Already at $CURRENT"
|
||||
echo "changed=false" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "Update available: $CURRENT -> $LATEST"
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
- name: Compute SRI hash of the new release
|
||||
id: hash
|
||||
if: steps.check.outputs.changed == 'true'
|
||||
env:
|
||||
LATEST: ${{ steps.latest.outputs.version }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
curl -sSL --fail \
|
||||
"https://github.com/yt-dlp/yt-dlp/releases/download/${LATEST}/yt-dlp" \
|
||||
-o /tmp/yt-dlp
|
||||
hash=$(nix hash file --sri --type sha256 /tmp/yt-dlp)
|
||||
case "$hash" in
|
||||
sha256-*) ;;
|
||||
*)
|
||||
echo "::error::Computed hash '${hash}' is not an SRI sha256 hash."
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
echo "SRI hash: $hash"
|
||||
echo "hash=${hash}" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Rewrite the yt-dlp pin in flake.nix
|
||||
if: steps.check.outputs.changed == 'true'
|
||||
env:
|
||||
CURRENT: ${{ steps.current.outputs.version }}
|
||||
LATEST: ${{ steps.latest.outputs.version }}
|
||||
HASH: ${{ steps.hash.outputs.hash }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
sed -i \
|
||||
-e "/ytDlp = pkgs.stdenv.mkDerivation/,/^ };\$/{ s|^\( *version = \"\)[^\"]*\(\";\)|\1${LATEST}\2|; }" \
|
||||
-e "/ytDlp = pkgs.stdenv.mkDerivation/,/^ };\$/{ s|\(url = \"https://github.com/yt-dlp/yt-dlp/releases/download/\)[^/]*\(/yt-dlp\";\)|\1${LATEST}\2|; }" \
|
||||
-e "/ytDlp = pkgs.stdenv.mkDerivation/,/^ };\$/{ s|^\( *hash = \"\)sha256-[^\"]*\(\";\)|\1${HASH}\2|; }" \
|
||||
flake.nix
|
||||
|
||||
echo "--- git diff --stat ---"
|
||||
git diff --stat flake.nix
|
||||
|
||||
changed_files=$(git diff --name-only)
|
||||
if [ "$changed_files" != "flake.nix" ]; then
|
||||
echo "::error::Expected only flake.nix to change, got: ${changed_files}"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
occurrences=$(grep -c "$LATEST" flake.nix || true)
|
||||
if [ "$occurrences" -ne 2 ]; then
|
||||
echo "::error::Expected the new version ${LATEST} to appear twice in flake.nix (version line + URL), found ${occurrences}."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if ! grep -q "$HASH" flake.nix; then
|
||||
echo "::error::New SRI hash was not written into flake.nix."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "Rewrote yt-dlp pin: ${CURRENT} -> ${LATEST}"
|
||||
|
||||
- name: Open pull request
|
||||
if: steps.check.outputs.changed == 'true'
|
||||
- name: Open PR if yt-dlp was updated
|
||||
if: steps.before.outputs.version != steps.after.outputs.version
|
||||
uses: peter-evans/create-pull-request@v6
|
||||
with:
|
||||
branch: auto/yt-dlp-update
|
||||
delete-branch: true
|
||||
commit-message: "chore(nix): yt-dlp ${{ steps.current.outputs.version }} → ${{ steps.latest.outputs.version }}"
|
||||
title: "chore(nix): yt-dlp ${{ steps.current.outputs.version }} → ${{ steps.latest.outputs.version }}"
|
||||
commit-message: "chore: yt-dlp ${{ steps.before.outputs.version }} → ${{ steps.after.outputs.version }}"
|
||||
title: "chore: yt-dlp ${{ steps.before.outputs.version }} → ${{ steps.after.outputs.version }}"
|
||||
body: |
|
||||
Automated bump of the pinned yt-dlp release. Old: `${{ steps.current.outputs.version }}`. New: `${{ steps.latest.outputs.version }}`. SRI hash: `${{ steps.hash.outputs.hash }}`.
|
||||
Automated `flake.lock` update. yt-dlp bumped from `${{ steps.before.outputs.version }}` to `${{ steps.after.outputs.version }}`.
|
||||
|
||||
Upstream release: https://github.com/yt-dlp/yt-dlp/releases/tag/${{ steps.latest.outputs.version }}
|
||||
Triggered by the weekly nixpkgs check.
|
||||
labels: dependencies
|
||||
|
|
|
|||
11
.gitignore
vendored
11
.gitignore
vendored
|
|
@ -9,17 +9,8 @@
|
|||
!docs/README*
|
||||
!docs/branding/
|
||||
!docs/branding/**
|
||||
# Dated design specs are tracked; docs/superpowers/plans/ stays ignored.
|
||||
!docs/superpowers/
|
||||
!docs/superpowers/specs/
|
||||
!docs/superpowers/specs/**
|
||||
!crates
|
||||
!crates/**
|
||||
# Static assets are built by Nix (frontendStatic derivation in flake.nix).
|
||||
# Do not commit them; `nix build` produces them from frontend/src/.
|
||||
crates/archivr-server/static/
|
||||
crates/archivr-server/static/**
|
||||
|
||||
|
||||
!vendor
|
||||
!vendor/**
|
||||
|
|
@ -51,3 +42,5 @@ frontend/result/**
|
|||
frontend/dist/
|
||||
frontend/dist/**
|
||||
.DS_Store
|
||||
frontend/storybook-static/
|
||||
frontend/storybook-static/**
|
||||
|
|
|
|||
129
AGENTS.md
129
AGENTS.md
|
|
@ -12,59 +12,13 @@ Three crates with a strict ownership split — **core owns truth; CLI and server
|
|||
|
||||
- `crates/archivr-core` — domain library: capture orchestration, SQLite schema/CRUD, downloaders, hashing. New archive features start here.
|
||||
- `crates/archivr-server` — Axum HTTP API + auth + static frontend serving.
|
||||
- `crates/archivr-cli` — clap-based CLI (`archivr` binary): `init`, `archive`, `yt-dlp status|update` subcommands.
|
||||
- `crates/archivr-cli` — clap-based CLI (`archivr` binary): `init`, `archive` subcommands.
|
||||
|
||||
Capture flow: locator → `determine_source()` (`crates/archivr-core/src/capture.rs`) routes by platform/shorthand (`yt:`, `x:`, `tweet:` …) → platform downloader (`downloader/ytdlp.rs`, `tweets.rs`, `singlefile.rs`, `http.rs`, `local.rs`) stages into `temp/` → SHA3-256 dedup (`hash.rs`, `downloader/store.rs`) moves blobs to `raw/A/B/HASH.EXT` → rows written to `archivr.sqlite` (runs, entries, artifacts, blobs) → served via `/api/archives/:id/...`. `CaptureConfig` carries per-request toggles (uBlock, reader mode, Freedium mirror, YouTube subtitles, etc.); when `via_freedium` is set, the fetch URL is rewritten through `freedium-mirror.cfd` while the canonical DB URL stays the original locator.
|
||||
Capture flow: locator → `determine_source()` (`crates/archivr-core/src/capture.rs`) routes by platform/shorthand (`yt:`, `x:`, `tweet:` …) → platform downloader (`downloader/ytdlp.rs`, `tweets.rs`, `singlefile.rs`, `http.rs`, `local.rs`) stages into `temp/` → SHA3-256 dedup (`hash.rs`, `downloader/store.rs`) moves blobs to `raw/A/B/HASH.EXT` → rows written to `archivr.sqlite` (runs, entries, artifacts, blobs) → served via `/api/archives/:id/...`. `CaptureConfig` carries per-request toggles (uBlock, reader mode, Freedium mirror, etc.); when `via_freedium` is set, the fetch URL is rewritten through `freedium-mirror.cfd` while the canonical DB URL stays the original locator.
|
||||
|
||||
YouTube playlists and channels produce a **parent container entry** with each video captured as a child entry. `downloader/ytdlp.rs` handles the flat-playlist probe (fetching per-video quality metadata before archiving); the multi-video download loop and sync mode (skipping already-archived videos when re-archiving a playlist or channel) live in `capture.rs`. Child order is persisted in `archived_entries.position` (append-only at insert in `database::create_archived_entry`; rewritten only by `database::reorder_child_entries` behind `PUT …/entries/:entry_uid/children/order` (allowed roles = `InstanceSettings::reorder_children_role_bits`, default ADMIN|OWNER, Owner-editable)).
|
||||
YouTube playlists and channels produce a **parent container entry** with each video captured as a child entry. `downloader/ytdlp.rs` handles the playlist probe (fetching per-video quality metadata before archiving), the multi-video download loop, and sync mode (skipping already-archived videos when re-archiving a playlist or channel).
|
||||
|
||||
YouTube videos (`Source::YouTubeVideo`: single videos and playlist/channel children; no other platform) also get up
|
||||
to two subtitle tracks (English + original language, manual over auto, VTT/SRT) in the same yt-dlp call by default.
|
||||
`CaptureConfig::download_subtitles` (manual `Default` = true) gates it; the API body field `download_subtitles` (absent
|
||||
= true), the CaptureDialog "Download subtitles" toggle and CLI `archivr archive --no-subtitles` feed it. Files are
|
||||
archived into `raw/` and registered as `subtitle` artifacts (`crates/archivr-core/src/subtitles.rs`). Subtitle
|
||||
failures are `eprintln!` warnings, never capture failures.
|
||||
|
||||
Pasted text takes a much shorter path: `perform_text_capture()` (`capture.rs`) skips source detection
|
||||
and every downloader shell-out — `downloader/text.rs` stages the body under `temp/`, hashes it, and the
|
||||
blob lands in `raw/` like any other artifact. Entrypoint is
|
||||
`POST /api/archives/:archive_id/captures/text`. Its body is byte-preserving, its entry has no fabricated
|
||||
`original_url`, and its normal text preview opens from the entry rail.
|
||||
|
||||
LLM summaries are a post-capture, manual-only subsystem: `crates/archivr-core/src/summarizer.rs` behind
|
||||
`GET`/`POST /api/archives/:archive_id/entries/:entry_uid/summary`, cached in the `entry_summaries`
|
||||
table per (entry, provider, model, prompt version, input hash). See `ARCHIVR-MENTAL-MODEL.md` for the
|
||||
provider set and the status lifecycle.
|
||||
|
||||
YouTube video entries are summarized from the best-ranked `subtitle` artifact reduced to a transcript, not the mp4.
|
||||
With no usable track the POST still returns 202: the row is created `pending` with the placeholder `input_sha256`
|
||||
`pending-subtitle-fetch`, and the background task fetches subtitles only (`build_summary_input_with_subtitle_fetch` →
|
||||
`subtitles::fetch_subtitles_for_entry`), writes the real hash, then runs the provider. The order is fixed (user
|
||||
requirement): archived subtitles → fetched subtitles → local transcription (`transcriber::transcribe_entry`, only when
|
||||
the POST body names a `transcribe_engine` and both earlier steps gave nothing) → error. If there are still none, the
|
||||
row fails with `NO_SUBTITLES_SUMMARY_MESSAGE` (or a transcription-specific copy when an engine ran) and no provider is
|
||||
called. Transcripts are `subtitle` artifacts with `kind: "transcribed"`, `origin: "transcription"`, `engine`, `model`.
|
||||
Spec + deviations: `docs/superpowers/specs/2026-10-05-local-transcription-fallback.md`.
|
||||
|
||||
X thread titles: `POST /api/archives/:archive_id/entries/:entry_uid/thread-title` (`ROLE_USER`, same gate as title
|
||||
PATCH; body `{provider}`) runs `thread_title::generate_thread_title` synchronously in one blocking task and saves
|
||||
`Thread about <topic> — @author` via `database::update_entry_title`. Never touches `entry_summaries`.
|
||||
|
||||
The requested provider model is the cache identity; a provider-returned resolved model is display attribution.
|
||||
At startup, pending/running attempts interrupted by shutdown are failed. A regeneration keeps the previous completed
|
||||
summary visible until its replacement completes; public readers receive completed content only, never diagnostics.
|
||||
|
||||
`SummaryBuildOptions` keeps summaries text-only unless `include_images` is set. The input digest includes that flag and
|
||||
the selected blobs' SHA-256, MIME types, and sizes, so a distinct image selection cannot reuse a text-only cache row.
|
||||
Candidates are `media` artifacts only: `jpg`/`jpeg`, `png`, `webp`, `gif`, and `avif`, capped at four images, 5 MiB each,
|
||||
and 12 MiB in aggregate. Anthropic HTTP, OpenAI-compatible HTTP, and Codex support images; Claude CLI does not. The core
|
||||
remains synchronous: the server puts provider work in its blocking boundary rather than introducing async to
|
||||
`archivr-core`.
|
||||
|
||||
Entry free-text search includes summary text (and generated JSON tags inside it) from the latest completed summary only.
|
||||
Pending and failed rows do not match, and a newer pending or failed request does not hide an older completed summary.
|
||||
|
||||
Per-archive layout (created by `archivr init`): `.archivr/` (name, store_path, `archivr.sqlite`) + sibling `store/` (`raw/`, `raw_tweets/`, `structured/`, `temp/`). Server-level auth lives in a **separate** `archivr-auth.sqlite` (users, sessions, API tokens, role bits GUEST=1/USER=2/ADMIN=4/OWNER=8; per-action role masks (e.g. `reorder_children_role_bits`) and per-provider thread-title models (`title_model_<kind>`) live on its `instance_settings` row).
|
||||
Per-archive layout (created by `archivr init`): `.archivr/` (name, store_path, `archivr.sqlite`) + sibling `store/` (`raw/`, `raw_tweets/`, `structured/`, `temp/`). Server-level auth lives in a **separate** `archivr-auth.sqlite` (users, sessions, API tokens, role bits GUEST=1/USER=2/ADMIN=4/OWNER=8).
|
||||
|
||||
The server mounts multiple archives from a TOML registry (`crates/archivr-server/src/registry.rs`); routes are parameterized by `:archive_id`.
|
||||
|
||||
|
|
@ -78,7 +32,6 @@ The server mounts multiple archives from a TOML registry (`crates/archivr-server
|
|||
| `frontend/src/` | React app: `App.jsx` (root state + custom routing), `api.js` (fetch client), `components/`, `styles.css` |
|
||||
| `docs/` | User docs (`README.md`), `superpowers/plans/` and `superpowers/specs/` (dated design docs — write plans there before large features) |
|
||||
| `modules/nixos/` | NixOS module (`services.archivr-server`) |
|
||||
| `.github/workflows/` | `update-ytdlp.yml` — weekly cron that PRs a yt-dlp version bump into `flake.nix` |
|
||||
| `vendor/twitter/` | Vendored Twitter scraper (active; the Python the server shells out to). Don't refactor casually. |
|
||||
| `vendor/readability/` | Mozilla `Readability.js`, concatenated into the SingleFile reader-mode browser script by `downloader/singlefile.rs`. |
|
||||
| `testing/` | Legacy scraping scripts + sample data. `testing/creds.txt` holds real tokens — never read, commit, or print it. |
|
||||
|
|
@ -95,13 +48,12 @@ cargo build --release -p archivr-server
|
|||
# Frontend (Bun, from frontend/)
|
||||
bun install
|
||||
bun run dev # Vite dev server
|
||||
bun run build # → ../crates/archivr-server/static (gitignored; only needed for bare cargo run)
|
||||
bun run build # → ../crates/archivr-server/static (served by the server)
|
||||
bun run storybook # Storybook on :6006
|
||||
|
||||
# Nix
|
||||
nix develop # devshell: yt-dlp, deno, nushell, uv, twitter-api-client
|
||||
nix build .#archivr-server # also .#archivr-cli, .#archivr-all; builds frontend automatically
|
||||
# If frontendDeps hash is stale (after bun.lock/package.json change):
|
||||
# nix build 2>&1 | grep "got:" → paste hash into flake.nix frontendDepsHash
|
||||
nix develop # devshell: yt-dlp, nushell, uv, twitter-api-client
|
||||
nix build .#archivr-server # also .#archivr-cli, .#archivr-all
|
||||
|
||||
# Docker
|
||||
docker compose up -d # port 8080; config in ./config/archivr-server.toml (see docker/config.example.toml)
|
||||
|
|
@ -119,49 +71,18 @@ No CI is configured; no rustfmt.toml/clippy.toml — default `cargo fmt`/`clippy
|
|||
- **State**: Axum `AppState { registry, auth_db_path, login_attempts }` via `State` extractor; middleware stack = `setup_guard` (503 until owner exists) → `login_rate_limit` (5/15min per IP) → `security_headers`.
|
||||
- **Auth extraction**: `AuthUser` implements `FromRequestParts` — session cookie (`session`) or `Authorization: Bearer` token (stored SHA3-256-hashed). Passwords are Argon2.
|
||||
- **Logging**: `eprintln!` with `info:`/`warn:` prefixes. No `tracing`/`log` — don't add structured logging piecemeal.
|
||||
- **External tools by env var**: `ARCHIVR_YT_DLP`, `ARCHIVR_DENO`, `ARCHIVR_JS_RUNTIME`, `ARCHIVR_CHROME`, `ARCHIVR_SINGLE_FILE`, `ARCHIVR_TWEET_PYTHON`, `ARCHIVR_TWEET_SCRAPER`, `ARCHIVR_STATIC_DIR`, `ARCHIVR_BIND`, `ARCHIVR_FFMPEG`. Local transcription: `ARCHIVR_TRANSCRIBE_ENGINES` (the gate; unset = off), `ARCHIVR_TRANSCRIBE_TIMEOUT` (default 3600s, whole job), `ARCHIVR_WHISPER_BACKEND` (`whisper_cpp`|`script`) / `ARCHIVR_WHISPER_CLI` / `ARCHIVR_WHISPER_MODEL` / `ARCHIVR_WHISPER_LANGUAGES`, `ARCHIVR_PARAKEET_CLI` / `ARCHIVR_PARAKEET_MODEL` / `ARCHIVR_PARAKEET_LANGUAGES`, `ARCHIVR_PHONON2_CLI` / `ARCHIVR_PHONON2_MODEL`. Downloaders shell out to subprocesses; resolve binaries through these vars. Env helpers (`required_env`, `env_or`, `optional_env`, `env_timeout`, `resolve_cli`) live in `env_config.rs`; bounded subprocesses go through `process::run_with_timeout` (yt-dlp calls keep `ytdlp.rs`'s private runner).
|
||||
- **yt-dlp is resolved, not just read**: `ARCHIVR_YT_DLP` (set by the flake wrappers) is only the
|
||||
*pinned candidate* handed to `resolve_yt_dlp()` (`downloader/ytdlp.rs`), which compares it against a
|
||||
self-updated copy in the state dir. `ARCHIVR_YT_DLP_FORCE` (absolute path) bypasses that comparison
|
||||
entirely; `archivr yt-dlp status` shows and chooses that forced candidate when it applies; `ARCHIVR_STATE_DIR`
|
||||
relocates the state dir. The JS runtime works the same way: `resolve_js_runtime()` (`downloader/js_runtime.rs`)
|
||||
picks the newest Deno ≥ 2.3.0 from `ARCHIVR_DENO` and `<state_dir>/deno/deno` (ties → state dir), then PATH;
|
||||
`ARCHIVR_JS_RUNTIME=RUNTIME[:ABS_PATH]` forces it. Both caches are `RwLock`s refreshed by `refresh_yt_dlp()` /
|
||||
`refresh_js_runtime()` after each successful component install (`ytdlp_tools::update_tools`), so a UI update needs
|
||||
no restart; `resolve_js_runtime()` returns an owned `Option<JsRuntime>`. Never cache a resolved path across
|
||||
operations. Never spawn bare yt-dlp — build every command with
|
||||
`yt_dlp_command()` so the resolver-chosen binary *and* `--js-runtimes` args are applied.
|
||||
- **LLM summaries by env var**: `ARCHIVR_ANTHROPIC_API_KEY` / `ARCHIVR_ANTHROPIC_URL` / `ARCHIVR_ANTHROPIC_MODEL`, `ARCHIVR_OPENAI_API_KEY` / `ARCHIVR_OPENAI_URL` / `ARCHIVR_OPENAI_MODEL`, `ARCHIVR_CLAUDE_CLI` / `ARCHIVR_CLAUDE_MODEL`, `ARCHIVR_CODEX_CLI` / `ARCHIVR_CODEX_MODEL`, plus `ARCHIVR_SUMMARY_HTTP_TIMEOUT` (default 120s) and `ARCHIVR_SUMMARY_CLI_TIMEOUT` (default 300s). Thread titles use `ARCHIVR_ANTHROPIC_TITLE_MODEL` (default `claude-haiku-4-5`), `ARCHIVR_OPENAI_TITLE_MODEL` (`gpt-4o-mini`), `ARCHIVR_CLAUDE_TITLE_MODEL` (`haiku`), `ARCHIVR_CODEX_TITLE_MODEL` (`gpt-6-luna`, override if unavailable) — never the summary model; `thread_title.rs` reuses provider transports via `summarizer::complete_plain`. The one exception to env-only config: admins may override the title model per provider in the auth DB (`instance_settings.title_model_{anthropic_http,openai_compatible,claude_cli,codex_cli}`; PATCH `/api/admin/instance-settings` trims, blank clears, rejects overlong or whitespace/control chars; GET returns `title_models.<kind>` with `model`/`source`/`fallback`). Precedence instance > `ARCHIVR_*_TITLE_MODEL` > default; the server passes the instance value to core as an `Option<&str>` override — core never reads the auth DB. Same convention as above — never TOML, which also keeps API keys out of anything the archive persists. Summaries are manual-only: nothing in `capture.rs` triggers them. The two CLI vars are optional overrides: unset, `resolve_cli()` auto-discovers well-known absolute installs first (`/opt/homebrew/bin/claude`, `/usr/local/bin/claude`; `/Applications/ChatGPT.app/Contents/Resources/codex`, `/opt/homebrew/bin/codex`, `/usr/local/bin/codex`), then `$HOME/.local/bin/<name>`, then the bare name on PATH — the absolute defaults matter because the ChatGPT desktop app ships `codex` off PATH. Frontend static output is generated; never hand-edit `crates/archivr-server/static/`.
|
||||
- **Frontend**: JSX (no TypeScript), PascalCase components in `frontend/src/components/`, kebab-case CSS classes, plain CSS with custom properties in `styles.css` (no Tailwind/CSS-in-JS). No router — `App.jsx` parses `window.location.pathname` + `history.pushState`. State = `useState` + one `AuthContext`; `sessionStorage` for refresh-resilient dialog state (see `CaptureDialog.jsx` job polling, 500ms). All API calls through `frontend/src/api.js` with relative `/api/*` URLs — add new endpoints there, not inline `fetch`. Summary polling and generate callbacks must remain scoped to the currently selected entry; image-inclusion behavior is unchanged. In-progress captures render through `SkeletonEntryRow.jsx`, which is a compact spinner + locator + "Archiving…" line (with a playlist/channel hint), **not** a grey skeleton block — don't reintroduce placeholder shimmer. Layout comes from semantic classes (e.g. `.capture-text-row` in `styles.css`), never from fallthrough on a generic row class; give a new row shape its own class.
|
||||
- **CLI providers parse a file, not stdout**: `codex` is invoked as `codex exec --output-last-message <tempfile> -` (prompt on stdin) and the reply is read back from that file — raw stdout carries a runtime header, an echo of the user prompt, and a `tokens used` footer that the JSON extractor will happily mistake for the answer. There is a positional-prompt fallback for older builds that reject `-`. Keep any new CLI provider on the same "give me only the final message" contract.
|
||||
- **Transcription engines follow the same rule**: whisper.cpp writes `-ovtt`; script engines (Whisper `script`, Parakeet) are run as `<script> --input <wav> --output <vtt> --model <m> [--language xx]` and must write WebVTT to `--output` (optional `<output>.lang`); the only stdout parsed is Phonon-2's documented `--json`, converted to VTT by `phonon_json_to_vtt`. Phonon-2 is English-only, hard-coded. One job at a time (process-wide slot).
|
||||
- **Log prefixes**: `info:`/`warn:` only. The one exception is the yt-dlp installer's `warning: python3 was not found on PATH …` (`ytdlp_tools.rs`), kept verbatim for byte-identical CLI stderr.
|
||||
- **External tools by env var**: `ARCHIVR_YT_DLP`, `ARCHIVR_CHROME`, `ARCHIVR_SINGLE_FILE`, `ARCHIVR_TWEET_PYTHON`, `ARCHIVR_TWEET_SCRAPER`, `ARCHIVR_STATIC_DIR`, `ARCHIVR_BIND`. Downloaders shell out to subprocesses; resolve binaries through these vars.
|
||||
- **Frontend**: JSX (no TypeScript), PascalCase components in `frontend/src/components/`, kebab-case CSS classes, plain CSS with custom properties in `styles.css` (no Tailwind/CSS-in-JS). No router — `App.jsx` parses `window.location.pathname` + `history.pushState`. State = `useState` + one `AuthContext`; `sessionStorage` for refresh-resilient dialog state (see `CaptureDialog.jsx` job polling, 500ms). All API calls through `frontend/src/api.js` with relative `/api/*` URLs — add new endpoints there, not inline `fetch`.
|
||||
- **Naming (Rust)**: standard snake_case/PascalCase; visibility and roles are bitflag `u32`s, not enums.
|
||||
|
||||
## Important Files
|
||||
|
||||
- `crates/archivr-server/src/main.rs` — server bootstrap: config load, archive mounting, auth DB init, stalled-job recovery (running → failed on startup), X Article title backfill (`capture::backfill_x_article_titles`; idempotent, startup only — CLI-only installs never run it).
|
||||
- `crates/archivr-server/src/main.rs` — server bootstrap: config load, archive mounting, auth DB init, stalled-job recovery (running → failed on startup).
|
||||
- `crates/archivr-server/src/routes.rs` — all HTTP handlers and the router; grep here first for API work.
|
||||
- `crates/archivr-core/src/capture.rs` — `perform_capture()`, `Source` enum, shorthand parsing; tweet titles prefer an X Article's `article.title` (`<title> — @handle`).
|
||||
- `crates/archivr-core/src/downloader/ytdlp.rs` — every yt-dlp shell-out (playlist/channel probe and download, sync mode, subtitle planning/args/staging, the combined media+subtitle call with its media-only retry, and the subtitles-only `download_subtitles`) **plus** the binary resolver: `resolve_yt_dlp()`, `state_dir()`, `probe_version()`.
|
||||
- `crates/archivr-core/src/subtitles.rs` — `subtitle` artifacts: archive/register (deduped per entry+blob), `subtitle_to_transcript()` (VTT/SRT reduction, rolling-caption dedup), `subtitle_track_rank()`, and summary-time `fetch_subtitles_for_entry()`.
|
||||
- `crates/archivr-core/src/downloader/text.rs` — pasted-text staging + hashing (`save()` → `StagedText`); accepts only `text/plain` and `text/markdown`.
|
||||
- `crates/archivr-core/src/summarizer.rs` — the `SummaryProvider` trait and its four implementations (Anthropic HTTP, OpenAI-compatible HTTP, `claude` CLI, `codex` CLI), `PROMPT_VERSION`, `resolve_cli()`, prompt assembly, `build_summary_input()` (artifact selection + HTML/text/JSON reduction; YouTube videos via subtitle transcript), `build_summary_input_with_subtitle_fetch()`, and the no-subtitles error/copy.
|
||||
- `crates/archivr-core/src/transcriber.rs` — local transcription: engine config from env (`transcriber_from_env`, `available_transcribers`), whisper.cpp/script/Phonon-2 adapters, audio acquisition + ffmpeg, the process-wide job slot, `transcribe_entry`, user-facing error copy.
|
||||
- `crates/archivr-core/src/process.rs` — `run_with_timeout` (kill on deadline, stderr tail) and the `ProcessTimedOut` sentinel; also backs summarizer `run_cli`.
|
||||
- `crates/archivr-core/src/env_config.rs` — shared `ARCHIVR_*` env helpers (`required_env`, `env_or`, `optional_env`, `env_timeout`, `resolve_cli`).
|
||||
- `crates/archivr-core/src/thread_title.rs` — thread-title prompt, topic sanitization, `Thread about … — @author` format, per-provider title model via `resolve_title_model` (precedence: instance setting `instance_settings.title_model_<kind>` passed in by the server > `ARCHIVR_*_TITLE_MODEL` > cheap default; core never reads the auth DB).
|
||||
- `crates/archivr-core/src/downloader/ytdlp_tools.rs` — yt-dlp + Deno update orchestration (`update_tools`, `install_yt_dlp`) and the status model (`tools_status`), shared by `archivr yt-dlp` and `/api/admin/yt-dlp[/update]`.
|
||||
- `crates/archivr-core/src/downloader/js_runtime.rs` — JS runtime for yt-dlp: `ARCHIVR_JS_RUNTIME` parsing, `DenoVersion`, `deno_candidates()`, `JsRuntimeRole` + `resolve_js_runtime_with_role()` (uncached; reports the winning slot so `status` stars exactly one row), `resolve_js_runtime()` (cached `RwLock`, owned clone, warns per resolution), `refresh_js_runtime()` and `js_runtime_args()`.
|
||||
- `crates/archivr-cli/src/main.rs` — CLI entry point; `archivr yt-dlp update|status` is a thin renderer over core `ytdlp_tools`.
|
||||
- `crates/archivr-core/src/downloader/deno_install.rs` — Deno release lookup, per-platform asset, staged + `--version`-verified atomic install into `<state_dir>/deno/deno`.
|
||||
- `.github/workflows/update-ytdlp.yml` — weekly (`0 6 * * 1`) + manual auto-bump of the `flake.nix` yt-dlp pin.
|
||||
- `crates/archivr-core/src/capture.rs` — `perform_capture()`, `Source` enum, shorthand parsing.
|
||||
- `crates/archivr-core/src/downloader/ytdlp.rs` — yt-dlp integration; YouTube playlist/channel probe and download, sync mode logic.
|
||||
- `crates/archivr-core/src/database.rs` — single source of truth for all SQLite schema and queries (both archive and auth DBs).
|
||||
- `frontend/src/App.jsx` / `frontend/src/api.js` — frontend root state and API surface.
|
||||
- `frontend/src/components/CaptureDialog.jsx` — `CaptureRow` (locator input, playlist quality selectors) and `CaptureTextRow` (the "Add text" flow), plus job polling.
|
||||
- `frontend/src/components/ContextRail.jsx` — the Summary section (provider selector, local-transcription engine selector, generate/regenerate, polling), the thread **Generate title** button, and the bulk panel's **Generate titles** (`handleBulkGenerateTitles`: ≥2 selected with ≥1 `tweet_thread`; skips non-threads; rail's Summary provider; per-entry `thread-title` calls 2 at a time; progress + `X updated, Y failed`; a selection change stops picking up new entries).
|
||||
- `frontend/src/components/SettingsView.jsx` — Settings, incl. the admin `YtDlpSection` (Instance › yt-dlp status + update).
|
||||
- `frontend/src/components/TextPreview.jsx` — preview renderer for text/markdown entries.
|
||||
- `docker/config.example.toml` — server config schema: `bind`, `auth_db_path`, repeated `[[archives]]` (`id`, `label`, `archive_path`).
|
||||
- `flake.nix`, `modules/nixos/archivr-server.nix`, `Dockerfile`, `docker-compose.yml` — deployment surfaces; config schema changes must be reflected in all of them plus `docs/README.md`.
|
||||
|
||||
|
|
@ -169,28 +90,12 @@ No CI is configured; no rustfmt.toml/clippy.toml — default `cargo fmt`/`clippy
|
|||
|
||||
- **Rust edition 2024** (root `Cargo.toml`); shared deps live in `[workspace.dependencies]` — add new deps there and reference with `workspace = true`.
|
||||
- **Bun** is the frontend package manager (`frontend/bun.lock`); use `bun`, not npm/yarn.
|
||||
- Runtime binaries the app expects on PATH or via env vars: `yt-dlp`, Deno (≥ 2.3.0, YouTube challenge solving), Chromium, `single-file` (Node), Python 3 with `twitter-api-client`, `ffmpeg`. `nix develop` provides the dev subset.
|
||||
- Runtime binaries the app expects on PATH or via env vars: `yt-dlp`, Chromium, `single-file` (Node), Python 3 with `twitter-api-client`, `ffmpeg`. `nix develop` provides the dev subset.
|
||||
- `.gitignore` is **default-deny with an allowlist** — new top-level files/dirs are invisible to git until explicitly allowed there.
|
||||
- Frontend build output (`crates/archivr-server/static/`) is generated by the `frontendStatic` Nix derivation; never hand-edit it, and do not commit it — it is excluded from git tracking. `nix build` is the standard workflow everywhere (local and NixOS) and builds the frontend automatically. `bun run build` only needed for bare `cargo run` one-off testing. When `bun.lock` or `package.json` changes, update `frontendDepsHash` in `flake.nix` for each system by running `nix build 2>&1 | grep "got:"` and pasting the reported hash.
|
||||
- **yt-dlp is pinned to a specific GitHub release** in the `ytDlp` derivation in `flake.nix` (zipapp
|
||||
fetched from `github.com/yt-dlp/yt-dlp/releases`, wrapped with `python312` + `ffmpeg`) — not taken
|
||||
from nixpkgs. Both the `archivr` and `archivr-server` wrappers set `ARCHIVR_YT_DLP` from it. Three
|
||||
ways to bump: the weekly `Update yt-dlp` workflow (automatic PR), `archivr yt-dlp update`
|
||||
(per-machine, into the state dir), or editing the three fields of the `ytDlp` block by hand. Which
|
||||
binary actually runs is decided at runtime by `resolve_yt_dlp()` in `downloader/ytdlp.rs`; use
|
||||
`archivr yt-dlp status` to see the candidates and the winner.
|
||||
- **Deno** comes from nixpkgs `pkgs.deno` in both wrappers (`ARCHIVR_DENO`, also on their PATH) and the devShell.
|
||||
The `Dockerfile` pins Deno 2.9.7 with a sha256 per arch — no auto-bump; change version and both hashes together.
|
||||
Docker installs yt-dlp via pip as `"yt-dlp[default]==<version>"`; the `[default]` extra pulls `yt-dlp-ejs`
|
||||
(the challenge solver) — dropping it brings back YouTube 403s even with Deno. The weekly `Update yt-dlp`
|
||||
workflow only bumps `flake.nix`, never the Dockerfile pin. Docker sets `ARCHIVR_STATE_DIR=/data/archivr-state`
|
||||
(on the persistent `/data` volume) so in-container `archivr yt-dlp update` survives restarts. On NixOS
|
||||
without `programs.nix-ld` the upstream Deno can't execute: `install_deno` (`archivr-core/src/downloader/deno_install.rs`)
|
||||
detects the spawn `NotFound`, skips, and keeps the pinned `ARCHIVR_DENO` if it is usable (≥ 2.3.0) —
|
||||
reported as ok, not a failure (exit 0 unless yt-dlp itself failed); with no usable pin it errors.
|
||||
- Frontend build output (`crates/archivr-server/static/`) is generated; never hand-edit it.
|
||||
|
||||
## Testing & QA
|
||||
|
||||
- **Rust**: unit tests only, in `#[cfg(test)]` modules inside source files (e.g. `capture.rs`, `database.rs`, `registry.rs`, `routes.rs`, `hash.rs`, and newer: `summarizer.rs` — 53 tests over provider construction, CLI/env resolution, output extraction, HTML/text/JSON input reduction, tweet-thread joining, YouTube subtitle selection/digest/no-subtitles errors and the transcription fallback order; `downloader/ytdlp.rs` — tests covering resolver priority and refresh, version tie-breaking, `yt_dlp_command_with` args, subtitle planning, argument construction, staging, the media-only retry decision and the timeout runner; `downloader/js_runtime.rs` — runtime spec parsing, Deno version parsing, candidate priority/ties/minimum version and the winning role, PATH fallback via `resolve_js_runtime_with_path` (never mutate the process `PATH`), refresh and `--js-runtimes` args; `downloader/deno_install.rs` — 7 tests: asset selection, release parsing and `verify_staged` (version mismatch, the "cannot execute" case); `downloader/ytdlp_tools.rs` — 2; `subtitles.rs` — 13 tests over VTT/SRT reduction, rolling-caption dedup, track ranking and artifact dedup; `transcriber.rs` — 40 (env config, argument builders, Phonon JSON → VTT against a real sample, language gating, end-to-end with fake engines); `process.rs` — 5; `env_config.rs` — 2; `thread_title.rs` — 8). Test locks: tests that set resolver env vars take `downloader::RESOLVER_ENV_LOCK` and call `refresh_*()` on cleanup so the cache never points into a deleted tempdir; tests that set transcription env vars or run `transcribe_entry` (including `summarizer.rs`'s fallback tests) take `transcriber::TRANSCRIBE_TEST_LOCK`; `summarizer.rs` and `env_config.rs` provider-env tests use their module-local `ENV_LOCK`. Tests that exec a freshly written script write it to a fresh path and wait out ETXTBSY first (`fake_deno`/`deno_install::tests::write_script` in core, `write_script` in the CLI). No `tests/` integration dir. Patterns: `tempfile` for scratch archives, config round-trip assertions, regex/parser validation. Run `cargo test` or `cargo test -p <crate>`.
|
||||
- **Frontend**: component render tests are colocated `*.test.jsx` files on `bun:test` (`bun test` from `frontend/`); there is no Storybook.
|
||||
- **Rust**: unit tests only, in `#[cfg(test)]` modules inside source files (e.g. `capture.rs`, `database.rs`, `registry.rs`, `routes.rs`, `hash.rs`). No `tests/` integration dir. Patterns: `tempfile` for scratch archives, config round-trip assertions, regex/parser validation. Run `cargo test` or `cargo test -p <crate>`.
|
||||
- **Frontend**: no test framework. Storybook (`bun run storybook`) is the component QA surface — stories are colocated `*.stories.jsx` files; add one when adding a nontrivial component.
|
||||
- Manual smoke test for server changes: build frontend, `cargo run -p archivr-server -- <config.toml>`, exercise `/api/*`.
|
||||
|
|
|
|||
|
|
@ -72,11 +72,8 @@ Rules:
|
|||
- Maximum nesting depth is 2 (root → child). Children cannot have children.
|
||||
- The UI shows child entries collapsed under their parent, expandable with a chevron.
|
||||
- Container entries are created by playlist/channel captures. Single-video and all other source types produce a standalone root entry with no children.
|
||||
- Children have a persisted sibling order: `archived_entries.position` (0-based per parent, `NULL` for roots). `list_child_entries` orders by it (tie-break `archived_at, id`).
|
||||
- New children are always appended (`MAX(position)+1` in `create_archived_entry`): an initial playlist/channel capture keeps playlist enumeration order; sync appends newly found videos after the existing (possibly user-reordered) children.
|
||||
- Users whose roles are allowed by the instance setting `reorder_children_role_bits` (auth DB `instance_settings`; default ADMIN|OWNER = 12; editable only by the Owner in Settings → Instance → Permissions) reorder an expanded parent's children on the main page. Exactly one control is shown, selected purely in CSS: the ↑/↓ buttons are the default, and a single `(hover: hover) and (pointer: fine) and (min-width: 641px)` query swaps in the drag handle (HTML5 drag-and-drop). Touch, no-pointer, phone-width (≤ 640px) viewports, and browsers that can't evaluate the query keep the arrows, since HTML5 DnD is unreliable there and no feature test detects it. Alt+↑/↓ on a focused child row works on all inputs (advertised via `aria-keyshortcuts`). The UI sends the full child UID list to `PUT /api/archives/:archive_id/entries/:entry_uid/children/order` (401 guest, 403 role not in mask, 404 unknown parent or a parent the caller can't see under the `list_child_entries` rule (`database::caller_sees_all_children`), 400 unless the list is exactly the current children). `/api/auth/me` and login return `can_reorder_children`; the UI hides reorder controls when it is false.
|
||||
|
||||
If a feature touches how entries are parented or how the UI groups them, start in `archivr-core` (`database.rs` for schema, `archive.rs` for listing, `capture.rs` for creation). Child-order UI lives in `routes.rs` (`reorder_entry_children_handler`) and `frontend/src/components/EntryRow.jsx`.
|
||||
If a feature touches how entries are parented or how the UI groups them, start in `archivr-core` (`database.rs` for schema, `archive.rs` for listing, `capture.rs` for creation).
|
||||
|
||||
## How To Run It
|
||||
|
||||
|
|
@ -84,7 +81,7 @@ There are two user-facing binaries:
|
|||
|
||||
| Binary | Purpose |
|
||||
|---|---|
|
||||
| `archivr` | CLI for initializing archives and capturing material into one archive (also `yt-dlp status\|update`; the web UI equivalent is Settings › Instance › yt-dlp) |
|
||||
| `archivr` | CLI for initializing archives and capturing material into one archive |
|
||||
| `archivr-server` | Web server for browsing one or more existing archives |
|
||||
|
||||
The CLI writes archive data:
|
||||
|
|
@ -131,27 +128,6 @@ sequenceDiagram
|
|||
CLI->>User: terminal result
|
||||
```
|
||||
|
||||
**YouTube subtitles are sidecar artifacts.** For `Source::YouTubeVideo` (single videos and YouTube playlist/channel
|
||||
children) with `CaptureConfig::download_subtitles` set (the default), the yt-dlp media call also writes up to two
|
||||
subtitle files (see [yt-dlp Lifecycle](#yt-dlp-lifecycle)). `subtitles::archive_staged_subtitles` moves them into
|
||||
`raw/` through `store::archive_staged_file` (same SHA3 dedup) before the temp dir is removed, and
|
||||
`subtitles::register_subtitle_artifacts` records each one as a `subtitle` artifact once the entry exists:
|
||||
`storage_area = "raw"`, MIME `text/vtt` or `application/x-subrip`, and `metadata_json`
|
||||
`{language, kind, format, original_language, origin}` with `kind` `manual`/`auto`/`unknown` and `origin` `capture` or
|
||||
`summary_fetch`. Registration runs in one `BEGIN IMMEDIATE` transaction and skips an existing
|
||||
(entry, `subtitle`, blob) row. Subtitle archive and register errors are warnings; the capture still succeeds. The CLI
|
||||
(`--no-subtitles`), the capture API (`download_subtitles: false`) and the capture dialog toggle all turn it off.
|
||||
|
||||
**Pasted text short-circuits most of that.** `perform_text_capture` (`capture.rs`) is not a `Source`
|
||||
route: there is no locator to classify, no URL probe, and no downloader subprocess. It validates the
|
||||
title (non-empty, ≤ 500 chars), the body (non-empty, ≤ 2 MiB) and the MIME type (`text/plain` or
|
||||
`text/markdown` only), then calls `downloader/text.rs` to write the bytes into `store/temp/<timestamp>/`
|
||||
and hash them. From there it rejoins the normal path — dedup into `raw/A/B/HASH.EXT`, then run, entry
|
||||
and artifact rows. The server exposes it as `POST /api/archives/:archive_id/captures/text`, and the UI
|
||||
drives it from `CaptureTextRow` in `CaptureDialog.jsx`. The body is preserved byte-for-byte, the entry's
|
||||
`original_url` remains empty (no fabricated `text:` URL), and its normal entry-rail preview renders through
|
||||
`TextPreview.jsx`.
|
||||
|
||||
## Web Capture Pipeline
|
||||
|
||||
Web pages (`Source::WebPage`) take a longer path than yt-dlp or tweets:
|
||||
|
|
@ -182,173 +158,6 @@ sequenceDiagram
|
|||
Server-->>Browser: JSON
|
||||
```
|
||||
|
||||
## LLM Summaries
|
||||
|
||||
Summaries are a **post-capture, manually triggered** subsystem. Nothing in `capture.rs` calls the
|
||||
summarizer; a summary exists only because someone pressed generate in the Summary section of
|
||||
`ContextRail.jsx`.
|
||||
|
||||
`crates/archivr-core/src/summarizer.rs` defines one `SummaryProvider` trait with four implementations:
|
||||
Anthropic HTTP, OpenAI-compatible HTTP, the local `claude` CLI, and the local `codex` CLI. Each is
|
||||
built purely from environment variables (`provider_from_env`), so no key or model name is ever written
|
||||
into archive data. `PROMPT_VERSION` in the same file stamps every row, so changing the prompt
|
||||
invalidates the cache instead of silently mixing generations.
|
||||
|
||||
`entry_summaries` (schema in `database.rs`) is that cache, unique on
|
||||
(`entry_id`, `provider_kind`, `provider_model`, `prompt_version`, `input_sha256`) — the requested provider
|
||||
model is the cache identity, so the same entry summarised by two providers, two requested models, or after a
|
||||
prompt change yields distinct rows, while a repeat request with identical inputs reuses one. When a provider
|
||||
returns its concrete resolved model, it is stored separately and displayed as attribution without changing that
|
||||
identity. Rows move `pending` → `running` → `completed` | `failed`, mirroring how capture jobs are tracked;
|
||||
the frontend polls only for the currently selected entry, and its generate callbacks are scoped to that same
|
||||
selection. Image-selection behavior remains unchanged.
|
||||
|
||||
On startup the server marks interrupted `pending` or `running` attempts failed. Regeneration is non-destructive:
|
||||
the prior completed summary stays visible until a replacement completes successfully. Public readers receive only
|
||||
completed summary content, never pending/failed state or diagnostic error text.
|
||||
|
||||
The summary path is deliberately explicit: UI consent (`Include attached images`) → core selection → input digest and
|
||||
cache lookup → provider transport → `pending`/`running`/`completed` lifecycle. Text is the default. When consent is
|
||||
present, `SummaryBuildOptions::include_images` admits only bounded `media` image candidates and the digest includes both
|
||||
the flag and selected blob identity, MIME type, and size. The core is still synchronous; the server owns the blocking
|
||||
boundary. Anthropic HTTP, OpenAI-compatible HTTP, and Codex can transport the selected image data; Claude CLI receives
|
||||
text only.
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
UI["ContextRail Summary"] -->|POST .../summary| Server
|
||||
Server --> Input["build_summary_input()"]
|
||||
Input --> Artifacts["entry artifacts on disk"]
|
||||
Server -->|YouTube, no usable subtitles| Fetch["fetch_subtitles_for_entry (yt-dlp)"]
|
||||
Fetch --> Artifacts
|
||||
Fetch -->|still none + transcribe_engine| Transcribe["transcribe_entry (audio → ffmpeg → engine)"]
|
||||
Transcribe --> Artifacts
|
||||
Server --> Row["entry_summaries: pending → running"]
|
||||
Server --> Provider["SummaryProvider (HTTP or CLI)"]
|
||||
Provider --> Row2["completed / failed"]
|
||||
UI -->|GET .../summary poll| Row2
|
||||
```
|
||||
|
||||
**X Articles and tweet threads are why the artifact lookup is special.** For most entries `build_summary_input` reads
|
||||
the single `primary_media` artifact. A `tweet` or `tweet_thread` entry has no `primary_media` — it has
|
||||
N `raw_tweet_json` artifacts, one per status in the thread. So the summarizer selects on the
|
||||
`raw_tweet_json` role instead, loads **all** matching artifacts in order, and joins them with
|
||||
`\n\n---\n\n`; a `---` line reads as a hard paragraph break to every model, keeping individual
|
||||
statuses from bleeding into one another. For an X Article, the reducer prefers article text over the tweet's body:
|
||||
`plain_text`, then flattened ordered blocks, then `preview_text`, then `summary_text`; only then does it fall back to
|
||||
`full_text`/`text`/`content`/`body`. Any change to article reduction or thread status storage must be mirrored here.
|
||||
|
||||
**Tweet entry titles.** When the status is an X Article, the title is `<article.title> — @handle`; otherwise the
|
||||
tweet-text excerpt (`caption_excerpt`). On server startup `capture::backfill_x_article_titles` retitles only rows whose title still
|
||||
byte-equals the legacy bare-link title recomputed from their raw JSON and whose raw JSON has an Article title. It is
|
||||
compare-and-set and idempotent. Titles carry no provenance flag, so exact equality is the guard: renamed titles are
|
||||
never touched, but a title renamed back to the exact legacy string is retitled. It runs only in `archivr-server`;
|
||||
CLI-only installs never backfill. Rearchive still does not change titles.
|
||||
|
||||
**Thread titles** (`thread_title.rs`) are manual and synchronous: the rail's **Generate title** button on
|
||||
`tweet_thread` entries calls `POST .../entries/:entry_uid/thread-title` (user role, the same gate as rename). It reuses
|
||||
the selected provider's transport (`summarizer::complete_plain`) with a cheap title model — admin instance setting
|
||||
`instance_settings.title_model_<kind>`, else `ARCHIVR_*_TITLE_MODEL`, else a per-provider default, never the summary
|
||||
model; the server passes the instance value to core as an `Option<&str>` override (core never reads the auth DB) — and
|
||||
never touches `entry_summaries`. The model returns only the topic; the server sanitizes it and owns the
|
||||
`Thread about <topic> — @author` format, saved via `update_entry_title`. The bulk panel's **Generate titles** (≥2
|
||||
selected, ≥1 thread) loops the same endpoint client-side over the selected threads, 2 at a time, with the rail's
|
||||
provider; a selection change stops it from starting further entries.
|
||||
|
||||
**YouTube videos are summarized from a `subtitle` artifact, never the mp4.** For `youtube`/`video` entries
|
||||
`build_summary_input` skips `primary_media` and reads the entry's `subtitle` artifacts (VTT/SRT by extension or MIME).
|
||||
`subtitles::subtitle_track_rank` orders them — 0 manual English, 1 manual original language, 2 other manual, 3
|
||||
transcribed (any language), 4 auto/unknown original language, 5 auto/unknown English, 6 the rest; ties go to the lowest
|
||||
artifact id — and the first
|
||||
track that reduces to a non-empty transcript wins. `subtitles::subtitle_to_transcript` drops header/`NOTE`/`STYLE`
|
||||
blocks, cue ids, timing lines, cue settings, inline tags and ASS overrides, decodes entities, and collapses rolling
|
||||
auto-caption repeats. Content is `Transcript ({language}, {kind} subtitles):` plus the transcript, truncated at
|
||||
48,000 chars like every input; the digest covers it, so adding or switching subtitles changes `input_sha256`.
|
||||
|
||||
No usable track yields `NoSubtitlesAvailable` (`is_no_subtitles_error`, distinct from unsupported content). The server
|
||||
preflight then inserts a `pending` row whose `input_sha256` is the placeholder `SUBTITLE_FETCH_PENDING_INPUT_SHA256`
|
||||
(`"pending-subtitle-fetch"`, never a real digest), skips the cache lookup, and returns 202. Its blocking task loads
|
||||
cookie rules from the auth DB and calls `build_summary_input_with_subtitle_fetch`: `fetch_subtitles_for_entry` returns
|
||||
early for non-YouTube entries, non-`http(s)` canonical URLs, or an entry that already has a usable track; otherwise it
|
||||
runs `fetch_metadata`, `plan_subtitle_request` and a subtitles-only `download_subtitles`, and registers the results
|
||||
with origin `summary_fetch`. An unreachable video or any yt-dlp error counts as zero subtitles. Then the input is
|
||||
rebuilt. On success `update_entry_summary_input_sha256` writes the real hash and the provider runs. If there are still
|
||||
no subtitles and the request named a `transcribe_engine`, `transcriber::transcribe_entry` runs (step 3 of the fixed
|
||||
order archived → fetched → transcribed → error), then the input is rebuilt once more. Transcription never runs when
|
||||
either earlier step yields a usable track. Otherwise the row fails with `NO_SUBTITLES_SUMMARY_MESSAGE` (or a
|
||||
transcription-specific copy when an engine ran) and no provider is called. Entries that already have usable subtitles
|
||||
keep the synchronous preflight.
|
||||
|
||||
**Local transcription** (`transcriber.rs`): engines `whisper` (whisper.cpp or a `script` wrapper), `parakeet` (script),
|
||||
`phonon2` (English only, `--json` stdout → VTT), configured only by env and gated by `ARCHIVR_TRANSCRIBE_ENGINES`
|
||||
(`GET /api/summary/transcription-engines` lists enabled ones). One job at a time per process; audio from the archived
|
||||
media or a yt-dlp audio download, ffmpeg to 16 kHz mono WAV, all within one `ARCHIVR_TRANSCRIBE_TIMEOUT` budget
|
||||
(ffmpeg and engines via `process::run_with_timeout`; the yt-dlp audio call via `ytdlp.rs`'s own runner). The result is a `subtitle` artifact with kind `transcribed`, origin `transcription`, plus
|
||||
`engine` and `model` metadata. Spec and deviations:
|
||||
`docs/superpowers/specs/2026-10-05-local-transcription-fallback.md`.
|
||||
|
||||
## yt-dlp Lifecycle
|
||||
|
||||
There is no single yt-dlp. Up to three can exist on one machine:
|
||||
|
||||
1. **The flake pin** — the `ytDlp` derivation in `flake.nix` fetches an exact release zipapp from
|
||||
`github.com/yt-dlp/yt-dlp/releases` and wraps it with `python312` + `ffmpeg`. Both the `archivr` and
|
||||
`archivr-server` wrappers export it as `ARCHIVR_YT_DLP`.
|
||||
2. **A state-dir install** — `archivr yt-dlp update` downloads the latest zipapp and installs it
|
||||
atomically (staged file, then rename) at `<state_dir>/yt-dlp/yt-dlp` with a sibling `.version`
|
||||
sentinel that lets repeat runs skip the download.
|
||||
3. **Whatever is on PATH** — the historical behaviour, and the last-resort fallback.
|
||||
|
||||
`resolve_yt_dlp()` in `downloader/ytdlp.rs` picks between them and caches the result in an `RwLock` until
|
||||
`refresh_yt_dlp()`: `ARCHIVR_YT_DLP_FORCE` wins outright if it points at a real file; otherwise the pinned and
|
||||
state-dir candidates are probed with `--version` and the newest wins — yt-dlp versions are `YYYY.MM.DD`,
|
||||
so plain string ordering is chronological — with exact ties going to the state-dir copy the user
|
||||
deliberately installed. If neither exists, it falls back to bare `yt-dlp`. `archivr yt-dlp status`
|
||||
prints every candidate, its version, and the winner; when the force variable applies, it includes that
|
||||
forced candidate and selects it as the winner.
|
||||
|
||||
Three ways to move the version forward: the weekly `.github/workflows/update-ytdlp.yml` cron (reads the
|
||||
current pin, queries the GitHub releases API, re-hashes with `nix hash file --sri`, rewrites the `ytDlp`
|
||||
block and opens a PR), `archivr yt-dlp update` / Settings › Instance › yt-dlp for one machine, or editing `flake.nix` by
|
||||
hand.
|
||||
|
||||
**The JS runtime has the same shape.** YouTube's player challenges are solved by yt-dlp's EJS solver, which needs
|
||||
Deno ≥ 2.3.0. Candidates: the Nix/Docker pin in `ARCHIVR_DENO`, `<state_dir>/deno/deno`, and `deno` on PATH.
|
||||
`resolve_js_runtime()` in `downloader/js_runtime.rs` (cached in an `RwLock` until `refresh_js_runtime()`, returns an owned clone, warnings printed once per resolution) returns a valid
|
||||
`ARCHIVR_JS_RUNTIME` force (`RUNTIME[:ABS_PATH]`, `deno|node|bun|quickjs`, invalid values warned and ignored)
|
||||
outright; otherwise it probes the pinned and state-dir Deno, drops anything below 2.3.0, compares real semver
|
||||
(`DenoVersion`, so 2.10.0 > 2.9.7) and keeps the newest, ties to the state dir; then PATH; else `None` plus a one-time
|
||||
warning. Only Deno is chosen automatically. Every yt-dlp process is built by `yt_dlp_command()` in `ytdlp.rs`, which
|
||||
appends `js_runtime_args()` (`--js-runtimes deno:<path>`; non-Deno forces get `--no-js-runtimes` first) — the
|
||||
`download` closure (incl. the media-only retry), `download_subtitles`, `fetch_metadata_with_timeout`,
|
||||
`fetch_playlist_info` and `probe_playlist_qualities`. The update also installs the latest Deno
|
||||
(`crates/archivr-core/src/downloader/deno_install.rs`): download, extract to `deno.new`, require `--version` to equal the release,
|
||||
then atomic rename. `status` adds a JS runtime table with rows `force (ARCHIVR_JS_RUNTIME)`, `env (ARCHIVR_DENO)`,
|
||||
`state-dir`, `path (deno)`; the star goes to the winning `JsRuntimeRole` from `resolve_js_runtime_with_role()`, not to
|
||||
every row whose path matches (the Nix wrappers' pinned Deno is also on PATH).
|
||||
|
||||
**One updater, two front ends.** `downloader/ytdlp_tools.rs` owns `install_yt_dlp`, `update_tools` and the
|
||||
`tools_status()` model; the CLI renders it as text and `GET /api/admin/yt-dlp` / `POST /api/admin/yt-dlp/update`
|
||||
(admin, 409 while an update runs) serve it to Settings › Instance › yt-dlp. `update_tools` calls `refresh_yt_dlp()` /
|
||||
`refresh_js_runtime()` after each successful component, so a UI update takes effect without a restart; commands
|
||||
already built keep their old binary. A CLI update runs in another process, so a running server still needs a restart.
|
||||
|
||||
**Subtitles ride on the media call.** When capture wants subtitles, `plan_subtitle_request` builds a bounded (≤ 2
|
||||
tracks) request from the `--dump-json` metadata capture already fetched. `download` then appends `--write-subs` and/or
|
||||
`--write-auto-subs` (only the kinds planned), `--sub-langs <codes>`, `--sub-format vtt/srt/best` and `--ignore-errors`.
|
||||
There is no `--convert-subs`, so ffmpeg is never needed for subtitles. yt-dlp treats `--sub-langs` entries as regexes,
|
||||
so planned codes are limited to `[A-Za-z0-9][A-Za-z0-9-]*`. Without metadata the request falls back to `en` and
|
||||
`.*-orig`. If the combined call exits non-zero without staging media and its stderr mentions subtitles, it is
|
||||
retried once with the exact legacy media-only arguments (`should_retry_media_only`), and subtitle files from the
|
||||
first attempt are still collected; other failures (private, deleted, geo-blocked) fail without a retry.
|
||||
`collect_staged_outputs` splits the staging dir into the
|
||||
media file and `<stem>.<lang>.<vtt|srt>` sidecars; other subtitle formats are dropped with a warning. Summary-time
|
||||
fetches use `download_subtitles` instead: `--skip-download --no-playlist --ignore-no-formats-error` plus the same
|
||||
subtitle args and `--ignore-errors`, staged under `temp/subs-<uuid>/`. A non-zero exit is tolerated if any subtitle
|
||||
file was written. The summary-time metadata probe and subtitle call are killed after `ARCHIVR_SUMMARY_CLI_TIMEOUT`
|
||||
(a timeout counts as "no subtitles"); capture-time calls stay unbounded. Blob cleanup refuses to run while such a
|
||||
fetch is in flight (`has_pending_subtitle_fetches`). Every one of these calls is built with `yt_dlp_command()`.
|
||||
|
||||
## Where To Edit
|
||||
|
||||
| Feature kind | Edit here |
|
||||
|
|
@ -358,21 +167,6 @@ fetch is in flight (`has_pending_subtitle_fetches`). Every one of these calls is
|
|||
| Archive opening, listing entries, entry detail, runs | `crates/archivr-core/src/archive.rs` |
|
||||
| Download/save behavior | `crates/archivr-core/src/downloader/` |
|
||||
| YouTube playlist/channel download, playlist probe, sync mode | `crates/archivr-core/src/downloader/ytdlp.rs` and `capture.rs` |
|
||||
| Which yt-dlp binary runs (resolver, state dir, version probe) | `crates/archivr-core/src/downloader/ytdlp.rs` |
|
||||
| Which JS runtime yt-dlp gets (Deno resolver, `ARCHIVR_JS_RUNTIME`, `--js-runtimes` args) | `crates/archivr-core/src/downloader/js_runtime.rs` |
|
||||
| yt-dlp/Deno update orchestration and status model (CLI + admin API) | `crates/archivr-core/src/downloader/ytdlp_tools.rs` |
|
||||
| Deno installer (CLI and UI update) | `crates/archivr-core/src/downloader/deno_install.rs` |
|
||||
| Settings › Instance › yt-dlp section | `frontend/src/components/SettingsView.jsx` (`YtDlpSection`) |
|
||||
| YouTube subtitles (track planning, yt-dlp args, staging) | `crates/archivr-core/src/downloader/ytdlp.rs` |
|
||||
| Subtitle artifacts, transcript reduction, track ranking, summary-time fetch | `crates/archivr-core/src/subtitles.rs` |
|
||||
| Pasted-text capture (staging, hashing, MIME allowlist) | `crates/archivr-core/src/downloader/text.rs` and `capture.rs` |
|
||||
| LLM summary providers, prompt, `PROMPT_VERSION`, input building | `crates/archivr-core/src/summarizer.rs` |
|
||||
| Local transcription engines, audio, job slot, `transcribe_entry` | `crates/archivr-core/src/transcriber.rs` |
|
||||
| Subprocess timeout runner | `crates/archivr-core/src/process.rs` |
|
||||
| Shared `ARCHIVR_*` env helpers | `crates/archivr-core/src/env_config.rs` |
|
||||
| Thread-title generation, cheap title models | `crates/archivr-core/src/thread_title.rs` |
|
||||
| X Article titles + startup backfill | `crates/archivr-core/src/capture.rs` (`backfill_x_article_titles`), called from `archivr-server/src/main.rs` |
|
||||
| `entry_summaries` schema and summary CRUD | `crates/archivr-core/src/database.rs` |
|
||||
| CLI commands, argument parsing, terminal output | `crates/archivr-cli/src/main.rs` |
|
||||
| Server API routes | `crates/archivr-server/src/routes.rs` |
|
||||
| Auth model (users, sessions, tokens, roles) | `crates/archivr-server/src/auth.rs` |
|
||||
|
|
@ -380,9 +174,6 @@ fetch is in flight (`has_pending_subtitle_fetches`). Every one of these calls is
|
|||
| Frontend root state + routing | `frontend/src/App.jsx` |
|
||||
| Frontend API client | `frontend/src/api.js` |
|
||||
| Frontend components | `frontend/src/components/` |
|
||||
| Summary UI (provider selector, transcription engine, generate, polling, thread Generate title) | `frontend/src/components/ContextRail.jsx` |
|
||||
| Text/Markdown entry preview | `frontend/src/components/TextPreview.jsx` |
|
||||
| "Add text" capture row | `frontend/src/components/CaptureDialog.jsx` |
|
||||
| Frontend styling | `frontend/src/styles.css` |
|
||||
|
||||
## Practical Feature Rule
|
||||
|
|
@ -403,10 +194,9 @@ If a browser feature needs new data, the usual order is:
|
|||
|
||||
The server both reads and writes archive data. Capture jobs are asynchronous: `POST /api/archives/:id/captures` inserts a job row, spawns a blocking task, and returns immediately; the frontend polls until the job completes or fails. Heavy work stays synchronous inside `archivr-core`.
|
||||
|
||||
**Auth model.** A separate `archivr-auth.sqlite` (path derived from the server config directory) holds users, sessions, and API tokens. Role bits are `u32` flags (`GUEST`, `USER`, `ADMIN`, `OWNER`) so a single bitmask value covers assignment, checks, and visibility. The middleware stack is `setup_guard` → `login_rate_limit` → `security_headers`; route families are classified `READ / ADMIN / WRITE / STATIC` in `routes.rs`. Configurable per-action permissions are `u32` role masks on the auth `instance_settings` singleton (currently `reorder_children_role_bits`). Handlers read them per request and allow when `caller_bits & mask != 0`, so changes take effect without re-login. Only the Owner may change them, and masks may contain only existing non-guest role bits (`database::grantable_role_bits`).
|
||||
**Auth model.** A separate `archivr-auth.sqlite` (path derived from the server config directory) holds users, sessions, and API tokens. Role bits are `u32` flags (`GUEST`, `USER`, `ADMIN`, `OWNER`) so a single bitmask value covers assignment, checks, and visibility. The middleware stack is `setup_guard` → `login_rate_limit` → `security_headers`; route families are classified `READ / ADMIN / WRITE / STATIC` in `routes.rs`.
|
||||
|
||||
**Search** is server-side free-text filtering over entry fields and the latest completed summary. The summary JSON is
|
||||
searched as text, so generated `tags` participate. Older completed summaries stay searchable while a newer request is
|
||||
pending or failed; rows with no completed summary contribute no summary-derived match.
|
||||
**Search** is client-side filtering over entries the frontend has already fetched.
|
||||
|
||||
**Admin view** covers mounted archives, users, sessions, and API tokens.
|
||||
|
||||
|
|
|
|||
48
Cargo.lock
generated
48
Cargo.lock
generated
|
|
@ -98,7 +98,7 @@ dependencies = [
|
|||
"clap",
|
||||
"regex",
|
||||
"rusqlite",
|
||||
"tempfile",
|
||||
"serde_json",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
|
|
@ -109,7 +109,6 @@ dependencies = [
|
|||
"base64",
|
||||
"chrono",
|
||||
"hex",
|
||||
"libc",
|
||||
"regex",
|
||||
"reqwest",
|
||||
"rusqlite",
|
||||
|
|
@ -118,7 +117,6 @@ dependencies = [
|
|||
"sha3",
|
||||
"tempfile",
|
||||
"uuid",
|
||||
"zip",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
|
|
@ -431,15 +429,6 @@ dependencies = [
|
|||
"libc",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "crc32fast"
|
||||
version = "1.5.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "01a7799fd6b852db0e61728dde9a204c423b44d689dbd432522543614b490e78"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "crypto-common"
|
||||
version = "0.1.6"
|
||||
|
|
@ -527,15 +516,6 @@ version = "0.1.4"
|
|||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "52051878f80a721bb68ebfbc930e07b65ba72f2da88968ea5c06fd6ca3d3a127"
|
||||
|
||||
[[package]]
|
||||
name = "flate2"
|
||||
version = "1.1.10"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "6e634e2e0ebac1ee034020da1ca582e17ffe4e0f5e985823721e168928136dcb"
|
||||
dependencies = [
|
||||
"zlib-rs",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "fnv"
|
||||
version = "1.0.7"
|
||||
|
|
@ -1562,7 +1542,6 @@ version = "1.0.150"
|
|||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e8014e44b4736ed0538adeecded0fce2a272f22dc9578a7eb6b2d9993c74cfb9"
|
||||
dependencies = [
|
||||
"indexmap",
|
||||
"itoa",
|
||||
"memchr",
|
||||
"serde",
|
||||
|
|
@ -1951,12 +1930,6 @@ version = "0.2.5"
|
|||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e421abadd41a4225275504ea4d6566923418b7f05506fbc9c0fe86ba7396114b"
|
||||
|
||||
[[package]]
|
||||
name = "typed-path"
|
||||
version = "0.12.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "8e28f89b80c87b8fb0cf04ab448d5dd0dd0ade2f8891bae878de66a75a28600e"
|
||||
|
||||
[[package]]
|
||||
name = "typenum"
|
||||
version = "1.19.0"
|
||||
|
|
@ -2493,25 +2466,6 @@ dependencies = [
|
|||
"syn",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "zip"
|
||||
version = "8.6.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "2d04a6b5381502aa6087c94c669499eb1602eb9c5e8198e534de571f7154809b"
|
||||
dependencies = [
|
||||
"crc32fast",
|
||||
"flate2",
|
||||
"indexmap",
|
||||
"memchr",
|
||||
"typed-path",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "zlib-rs"
|
||||
version = "0.6.8"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b268e58e7c693d7c271f93ffc4ba3b380412554231c85bf61ca7af91042a4112"
|
||||
|
||||
[[package]]
|
||||
name = "zmij"
|
||||
version = "1.0.21"
|
||||
|
|
|
|||
|
|
@ -19,7 +19,7 @@ hex = "0.4.3"
|
|||
regex = "1.12.2"
|
||||
rusqlite = { version = "0.32.1", features = ["bundled"] }
|
||||
serde = { version = "1.0.228", features = ["derive"] }
|
||||
serde_json = { version = "1.0.132", features = ["preserve_order"] }
|
||||
serde_json = "1.0.132"
|
||||
sha3 = "0.10.8"
|
||||
tempfile = "3.13.0"
|
||||
tokio = { version = "1.41.1", features = ["macros", "rt-multi-thread", "net", "fs", "io-util"] }
|
||||
|
|
@ -33,5 +33,3 @@ argon2 = { version = "0.5", features = ["std"] }
|
|||
rand = { version = "0.8", features = ["std"] }
|
||||
axum-extra = { version = "0.9", features = ["cookie"] }
|
||||
parking_lot = "0.12"
|
||||
zip = { version = "8.6", default-features = false, features = ["deflate-flate2-zlib-rs"] }
|
||||
libc = "0.2"
|
||||
|
|
|
|||
39
Dockerfile
39
Dockerfile
|
|
@ -42,8 +42,6 @@ RUN touch \
|
|||
###############################################################################
|
||||
FROM debian:bookworm-slim
|
||||
|
||||
ARG TARGETARCH
|
||||
|
||||
# Runtime dependencies:
|
||||
# chromium used by single-file-cli for full-page archiving
|
||||
# nodejs (20+) runtime for single-file-cli (requires Node >=20; Debian
|
||||
|
|
@ -53,10 +51,6 @@ ARG TARGETARCH
|
|||
# python3 + pip + venv twitter scraper
|
||||
# ca-certificates outbound HTTPS from the server and NodeSource HTTPS
|
||||
# libssl3 OpenSSL linked by the Rust binary
|
||||
# unzip unpacks the Chromium extensions and the Deno release zip
|
||||
# deno (pinned, below) JS runtime for yt-dlp's YouTube challenge solver. Node 20
|
||||
# is not usable: yt-dlp needs Node >= 22 and enables only
|
||||
# Deno by default.
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
curl \
|
||||
ca-certificates \
|
||||
|
|
@ -77,36 +71,10 @@ RUN npm install -g single-file-cli
|
|||
|
||||
# Install yt-dlp and twitter-api-client into an isolated venv to avoid
|
||||
# conflicts with Debian's system Python packages.
|
||||
# The [default] extra pulls in yt-dlp-ejs (the EJS challenge-solver library);
|
||||
# without it Deno alone cannot solve YouTube challenges.
|
||||
RUN python3 -m venv /opt/archivr-venv \
|
||||
&& /opt/archivr-venv/bin/pip install --no-cache-dir \
|
||||
"yt-dlp[default]==2026.8.19" \
|
||||
"twitter-api-client==0.10.22"
|
||||
|
||||
# Pinned Deno — fallback JS runtime for yt-dlp's YouTube challenge solver.
|
||||
# A newer copy installed by `archivr yt-dlp update` into ARCHIVR_STATE_DIR
|
||||
# takes precedence. No auto-bump workflow exists: bump DENO_VERSION and BOTH
|
||||
# sha256 values together. Adds roughly 80 MB to the image.
|
||||
RUN set -eu; \
|
||||
DENO_VERSION=2.9.7; \
|
||||
DENO_SHA256_AMD64=c6527f24f4b16031d3ae4fa9f658d5f11534c8d84ce7dc8502420280919c3490; \
|
||||
DENO_SHA256_ARM64=c832298b1ad4422481334855f6003e0f54145762c5a134f20a489511d2f65bbf; \
|
||||
arch="${TARGETARCH:-$(dpkg --print-architecture)}"; \
|
||||
case "$arch" in \
|
||||
amd64) triple=x86_64-unknown-linux-gnu; sha="$DENO_SHA256_AMD64" ;; \
|
||||
arm64) triple=aarch64-unknown-linux-gnu; sha="$DENO_SHA256_ARM64" ;; \
|
||||
*) echo "ERROR: unsupported architecture for Deno: $arch"; exit 1 ;; \
|
||||
esac; \
|
||||
command -v unzip >/dev/null; \
|
||||
curl -fsSL "https://github.com/denoland/deno/releases/download/v${DENO_VERSION}/deno-${triple}.zip" \
|
||||
-o /tmp/deno.zip; \
|
||||
echo "${sha} /tmp/deno.zip" | sha256sum -c -; \
|
||||
mkdir -p /usr/local/lib/archivr/deno; \
|
||||
unzip -q -o /tmp/deno.zip deno -d /usr/local/lib/archivr/deno; \
|
||||
rm /tmp/deno.zip; \
|
||||
chmod 0755 /usr/local/lib/archivr/deno/deno; \
|
||||
/usr/local/lib/archivr/deno/deno --version
|
||||
yt-dlp \
|
||||
twitter-api-client
|
||||
|
||||
# Download Chromium extensions used during headless captures.
|
||||
# uBlock Origin Lite (MV3) — ad/tracker blocking.
|
||||
|
|
@ -149,9 +117,6 @@ ENV ARCHIVR_STATIC_DIR=/usr/share/archivr-server/static \
|
|||
ARCHIVR_TWEET_PYTHON=/opt/archivr-venv/bin/python3 \
|
||||
ARCHIVR_TWEET_SCRAPER=/usr/local/lib/archivr/scrape_user_tweet_contents.py \
|
||||
ARCHIVR_YT_DLP=/opt/archivr-venv/bin/yt-dlp \
|
||||
ARCHIVR_DENO=/usr/local/lib/archivr/deno/deno \
|
||||
ARCHIVR_FFMPEG=/usr/bin/ffmpeg \
|
||||
ARCHIVR_STATE_DIR=/data/archivr-state \
|
||||
ARCHIVR_UBLOCK_EXT=/usr/local/lib/archivr/extensions/ublock-origin-lite \
|
||||
ARCHIVR_COOKIE_EXT=/usr/local/lib/archivr/extensions/istilldontcareaboutcookies \
|
||||
ARCHIVR_CHROME_ARGS=--no-sandbox
|
||||
|
|
|
|||
|
|
@ -14,6 +14,4 @@ chrono.workspace = true
|
|||
clap.workspace = true
|
||||
regex.workspace = true
|
||||
rusqlite.workspace = true
|
||||
|
||||
[dev-dependencies]
|
||||
tempfile.workspace = true
|
||||
serde_json.workspace = true
|
||||
|
|
|
|||
|
|
@ -1,11 +1,11 @@
|
|||
use anyhow::{bail, Context, Result};
|
||||
use archivr_core::{
|
||||
archive,
|
||||
capture::CaptureConfig,
|
||||
downloader::ytdlp_tools::{tools_status, update_tools, ToolCandidate},
|
||||
};
|
||||
use anyhow::{Context, Result};
|
||||
use archivr_core::{archive, capture::CaptureConfig};
|
||||
use clap::{Parser, Subcommand};
|
||||
use std::{env, path::Path, process};
|
||||
use std::{
|
||||
env,
|
||||
path::Path,
|
||||
process,
|
||||
};
|
||||
|
||||
#[derive(Parser, Debug)]
|
||||
#[command(version, about, long_about = None)]
|
||||
|
|
@ -20,9 +20,6 @@ enum Command {
|
|||
Archive {
|
||||
/// URL or Path to archive
|
||||
path: String,
|
||||
/// Skip downloading YouTube subtitles
|
||||
#[arg(long)]
|
||||
no_subtitles: bool,
|
||||
},
|
||||
Init {
|
||||
/// Path to initialize the archive in
|
||||
|
|
@ -51,33 +48,13 @@ enum Command {
|
|||
#[arg(long = "force-with-info-removal")]
|
||||
force_with_info_removal: bool,
|
||||
},
|
||||
|
||||
/// Inspect or update the yt-dlp binary and the JavaScript runtime (Deno) archivr runs
|
||||
#[command(name = "yt-dlp")]
|
||||
YtDlp {
|
||||
#[command(subcommand)]
|
||||
subcmd: YtDlpCmd,
|
||||
},
|
||||
}
|
||||
|
||||
#[derive(Subcommand, Debug)]
|
||||
enum YtDlpCmd {
|
||||
/// Download the latest yt-dlp zipapp and Deno into archivr's state directory
|
||||
Update {
|
||||
/// Install this exact yt-dlp release tag instead of the latest (e.g. 2026.09.15);
|
||||
/// applies to yt-dlp only — Deno always installs the latest release
|
||||
#[arg(long)]
|
||||
version: Option<String>,
|
||||
},
|
||||
/// Show every yt-dlp and JS runtime candidate, its version, and which one wins
|
||||
Status,
|
||||
}
|
||||
|
||||
fn main() -> Result<()> {
|
||||
let args = Args::parse();
|
||||
|
||||
match args.command {
|
||||
Command::Archive { ref path, no_subtitles } => {
|
||||
Command::Archive { ref path } => {
|
||||
let archive_path = match archive::find_archive_path()? {
|
||||
Some(path) => path,
|
||||
None => {
|
||||
|
|
@ -86,11 +63,7 @@ fn main() -> Result<()> {
|
|||
}
|
||||
};
|
||||
let archive_paths = archive::read_archive_paths(&archive_path)?;
|
||||
let config = CaptureConfig {
|
||||
download_subtitles: !no_subtitles,
|
||||
..CaptureConfig::default()
|
||||
};
|
||||
let result = archivr_core::capture::perform_capture(&archive_paths, path, None, None, &config)?;
|
||||
let result = archivr_core::capture::perform_capture(&archive_paths, path, None, None, &CaptureConfig::default())?;
|
||||
println!("Archived: run {}", result.run_uid);
|
||||
Ok(())
|
||||
}
|
||||
|
|
@ -123,205 +96,7 @@ fn main() -> Result<()> {
|
|||
);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
Command::YtDlp { subcmd } => match subcmd {
|
||||
YtDlpCmd::Update { version } => yt_dlp_update(version.as_deref()),
|
||||
YtDlpCmd::Status => yt_dlp_status(),
|
||||
},
|
||||
} // _ => eprintln!("Unknown command: {:?}", args.command),
|
||||
}
|
||||
}
|
||||
|
||||
/// Formats one `status` row: `role\tlocation\tversion\tchosen`. Missing candidates and
|
||||
/// unknown versions show an em dash.
|
||||
fn format_status_row(
|
||||
role: &str,
|
||||
location: Option<&str>,
|
||||
version: Option<&str>,
|
||||
chosen: bool,
|
||||
) -> String {
|
||||
match location {
|
||||
Some(loc) => {
|
||||
let version = version.unwrap_or("—");
|
||||
let star = if chosen { "*" } else { "" };
|
||||
format!("{role}\t{loc}\t{version}\t{star}")
|
||||
}
|
||||
None => format!("{role}\t—\t—\t"),
|
||||
}
|
||||
}
|
||||
|
||||
/// Renders one candidate row; an invalid override or a candidate whose `--version`
|
||||
/// probe fails shows its reason in the version column.
|
||||
fn candidate_row(c: &ToolCandidate) -> String {
|
||||
match &c.invalid {
|
||||
Some(reason) => format_status_row(
|
||||
c.label,
|
||||
c.path.as_deref(),
|
||||
Some(&format!("invalid: {reason}")),
|
||||
c.chosen,
|
||||
),
|
||||
None => format_status_row(c.label, c.path.as_deref(), c.version.as_deref(), c.chosen),
|
||||
}
|
||||
}
|
||||
|
||||
fn yt_dlp_status() -> Result<()> {
|
||||
let s = tools_status();
|
||||
|
||||
println!("role\tpath\tversion\tchosen");
|
||||
for c in &s.yt_dlp {
|
||||
println!("{}", candidate_row(c));
|
||||
}
|
||||
if let Some(target) = s.yt_dlp_target.as_deref().filter(|_| !s.yt_dlp_installed) {
|
||||
println!("\nNo state-dir install yet; `archivr yt-dlp update` would write to {target}");
|
||||
}
|
||||
|
||||
println!("\nJS runtime (passed to yt-dlp as --js-runtimes)");
|
||||
println!("role\tpath\tversion\tchosen");
|
||||
for c in &s.js_runtime {
|
||||
println!("{}", candidate_row(c));
|
||||
}
|
||||
if s.js_runtime_chosen.is_none() {
|
||||
println!(
|
||||
"\nNo JS runtime resolved — YouTube downloads may fail with HTTP 403; run `archivr yt-dlp update`"
|
||||
);
|
||||
}
|
||||
if let Some(slot) = s.deno_target.as_deref().filter(|_| !s.deno_installed) {
|
||||
println!("\nNo state-dir deno yet; `archivr yt-dlp update` would write to {slot}");
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Installs yt-dlp and Deno independently: a Deno failure never blocks the yt-dlp
|
||||
/// update (and vice versa). Both outcomes are reported; any failure exits non-zero.
|
||||
fn yt_dlp_update(requested_version: Option<&str>) -> Result<()> {
|
||||
let report = update_tools(
|
||||
requested_version,
|
||||
concat!("archivr-cli/", env!("CARGO_PKG_VERSION")),
|
||||
false,
|
||||
&mut |l| println!("{l}"),
|
||||
)?;
|
||||
|
||||
println!("\nSummary:");
|
||||
match &report.yt_dlp {
|
||||
Ok(_) => println!(" yt-dlp: ok"),
|
||||
Err(e) => println!(" yt-dlp: FAILED: {e:#}"),
|
||||
}
|
||||
match &report.deno {
|
||||
Ok(msg) => println!(" deno: {msg}"),
|
||||
Err(e) => println!(" deno: FAILED: {e:#}"),
|
||||
}
|
||||
|
||||
let failed = report.failed_components();
|
||||
if !failed.is_empty() {
|
||||
bail!("update failed for: {}", failed.join(", "));
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::{candidate_row, format_status_row};
|
||||
use archivr_core::downloader::ytdlp_tools::ToolCandidate;
|
||||
use archivr_core::downloader::ytdlp::{
|
||||
forced_yt_dlp, probe_version, resolve_yt_dlp_uncached, YT_DLP_FORCE_ENV,
|
||||
};
|
||||
use std::path::Path;
|
||||
|
||||
/// Writes an executable script to `path` (callers must use a fresh path each time), then
|
||||
/// waits until it can be exec'd. A child forked by a parallel test while our write fd was
|
||||
/// open keeps a copy of it until that child execs, so our own exec can fail with ETXTBSY
|
||||
/// (rust-lang/rust#114554). One exec that isn't ETXTBSY proves no writer is left, and none
|
||||
/// can appear later because our fd is already closed.
|
||||
#[cfg(unix)]
|
||||
fn write_script(path: &Path, body: &str) {
|
||||
use std::os::unix::fs::PermissionsExt;
|
||||
std::fs::create_dir_all(path.parent().unwrap()).unwrap();
|
||||
std::fs::write(path, body).unwrap();
|
||||
std::fs::set_permissions(path, std::fs::Permissions::from_mode(0o755)).unwrap();
|
||||
for _ in 0..200 {
|
||||
match std::process::Command::new(path).arg("--version").output() {
|
||||
Err(e) if e.kind() == std::io::ErrorKind::ExecutableFileBusy => {
|
||||
std::thread::sleep(std::time::Duration::from_millis(5));
|
||||
}
|
||||
_ => return,
|
||||
}
|
||||
}
|
||||
panic!("{} stayed busy (ETXTBSY)", path.display());
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
#[test]
|
||||
fn forced_candidate_is_rendered_and_selected() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let forced = tmp.path().join("forced/yt-dlp");
|
||||
write_script(&forced, "#!/bin/sh\necho 2020.01.01\n");
|
||||
unsafe { std::env::set_var(YT_DLP_FORCE_ENV, &forced) };
|
||||
|
||||
let candidate = forced_yt_dlp();
|
||||
assert_eq!(candidate.as_deref(), Some(forced.as_path()));
|
||||
let chosen = resolve_yt_dlp_uncached();
|
||||
assert_eq!(chosen, forced);
|
||||
let version = probe_version(&forced);
|
||||
assert_eq!(
|
||||
format_status_row(
|
||||
"force (ARCHIVR_YT_DLP_FORCE)",
|
||||
Some(&forced.display().to_string()),
|
||||
version.as_deref(),
|
||||
chosen == forced,
|
||||
),
|
||||
format!(
|
||||
"force (ARCHIVR_YT_DLP_FORCE)\t{}\t2020.01.01\t*",
|
||||
forced.display()
|
||||
)
|
||||
);
|
||||
|
||||
unsafe { std::env::remove_var(YT_DLP_FORCE_ENV) };
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn missing_candidate_renders_dashes() {
|
||||
assert_eq!(
|
||||
format_status_row("state-dir", None, Some("2.9.7"), true),
|
||||
"state-dir\t—\t—\t"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn unknown_version_renders_dash() {
|
||||
assert_eq!(
|
||||
format_status_row("path (deno)", Some("/bin/deno"), None, false),
|
||||
"path (deno)\t/bin/deno\t—\t"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn invalid_override_row_shows_reason_and_is_not_chosen() {
|
||||
assert_eq!(
|
||||
format_status_row(
|
||||
"force (ARCHIVR_JS_RUNTIME)",
|
||||
Some("python"),
|
||||
Some("invalid: unknown runtime python (expected deno, node, bun or quickjs)"),
|
||||
false,
|
||||
),
|
||||
"force (ARCHIVR_JS_RUNTIME)\tpython\tinvalid: unknown runtime python \
|
||||
(expected deno, node, bun or quickjs)\t"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn candidate_row_renders_invalid_override() {
|
||||
let c = ToolCandidate {
|
||||
role: "force",
|
||||
label: "force (ARCHIVR_JS_RUNTIME)",
|
||||
path: Some("python".into()),
|
||||
version: None,
|
||||
chosen: false,
|
||||
invalid: Some("unknown runtime python (expected deno, node, bun or quickjs)".into()),
|
||||
};
|
||||
assert_eq!(
|
||||
candidate_row(&c),
|
||||
"force (ARCHIVR_JS_RUNTIME)\tpython\tinvalid: unknown runtime python \
|
||||
(expected deno, node, bun or quickjs)\t"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -15,10 +15,6 @@ sha3.workspace = true
|
|||
uuid.workspace = true
|
||||
reqwest = { workspace = true }
|
||||
base64.workspace = true
|
||||
zip.workspace = true
|
||||
|
||||
[target.'cfg(unix)'.dependencies]
|
||||
libc.workspace = true
|
||||
|
||||
[dev-dependencies]
|
||||
tempfile = "3"
|
||||
|
|
|
|||
|
|
@ -36,14 +36,6 @@ pub struct EntrySummary {
|
|||
pub cacheable_bytes: i64,
|
||||
}
|
||||
|
||||
/// One stored LLM summary, as exposed over the API.
|
||||
///
|
||||
/// Aliased rather than redefined: the DB row is already the exact shape the
|
||||
/// frontend needs, and a second near-identical struct would only add a mapping
|
||||
/// step to keep in sync. The `View` name exists because `EntrySummary` in this
|
||||
/// module is the *entry listing* row, an unrelated thing.
|
||||
pub use crate::database::EntrySummaryRecord as EntrySummaryView;
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq, serde::Serialize)]
|
||||
pub struct EntryDetail {
|
||||
pub summary: EntrySummary,
|
||||
|
|
@ -51,12 +43,6 @@ pub struct EntryDetail {
|
|||
pub source_metadata_json: String,
|
||||
pub display_metadata_json: Option<String>,
|
||||
pub artifacts: Vec<EntryArtifactSummary>,
|
||||
/// Most recent completed summary for this entry. Always `None` on a fresh
|
||||
/// capture — summarization is manual.
|
||||
pub latest_summary: Option<EntrySummaryView>,
|
||||
/// Latest non-completed generation attempt, kept separate so a replacement
|
||||
/// never displaces readable completed content.
|
||||
pub summary_attempt: Option<EntrySummaryView>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq, serde::Serialize)]
|
||||
|
|
@ -357,17 +343,12 @@ pub fn get_entry_detail(
|
|||
})?
|
||||
.collect::<rusqlite::Result<Vec<_>>>()?;
|
||||
|
||||
let latest_summary = database::latest_completed_entry_summary(conn, entry_id)?;
|
||||
let summary_attempt = database::latest_entry_summary_attempt(conn, entry_id)?;
|
||||
|
||||
Ok(Some(EntryDetail {
|
||||
summary,
|
||||
structured_root_relpath,
|
||||
source_metadata_json,
|
||||
display_metadata_json,
|
||||
artifacts,
|
||||
latest_summary,
|
||||
summary_attempt,
|
||||
}))
|
||||
}
|
||||
|
||||
|
|
@ -510,11 +491,9 @@ pub fn list_entries_for_collection(
|
|||
Ok(entries)
|
||||
}
|
||||
|
||||
/// Returns the direct children of the entry identified by `parent_entry_uid`
|
||||
/// in persisted sibling order (`position`, set at insert time as append and
|
||||
/// rewritten by `database::reorder_child_entries`), tie-broken by
|
||||
/// `archived_at, id`. Returns an empty vec if the parent has no children or
|
||||
/// does not exist.
|
||||
/// Returns the direct children of the entry identified by `parent_entry_uid`,
|
||||
/// ordered ascending by `archived_at, id` (preserves playlist ordinal feel).
|
||||
/// Returns an empty vec if the parent has no children or does not exist.
|
||||
pub fn list_child_entries(
|
||||
conn: &rusqlite::Connection,
|
||||
parent_entry_uid: &str,
|
||||
|
|
@ -537,7 +516,7 @@ pub fn list_child_entries(
|
|||
)\
|
||||
) \
|
||||
GROUP BY e.id \
|
||||
ORDER BY e.position ASC, e.archived_at ASC, e.id ASC",
|
||||
ORDER BY e.archived_at ASC, e.id ASC",
|
||||
ENTRY_SELECT_COLS, ENTRY_FROM_JOINS,
|
||||
);
|
||||
let mut stmt = conn.prepare(&sql)?;
|
||||
|
|
@ -780,14 +759,7 @@ pub fn search_entries(
|
|||
sql.push_str(&format!(
|
||||
" AND (LOWER(e.title) LIKE ?{n} OR LOWER(si.canonical_url) LIKE ?{n} \
|
||||
OR LOWER(e.entry_uid) LIKE ?{n} OR LOWER(e.source_kind) LIKE ?{n} \
|
||||
OR LOWER(e.entity_kind) LIKE ?{n} OR LOWER(e.visibility) LIKE ?{n} \
|
||||
OR LOWER(COALESCE((\
|
||||
SELECT s.summary_text FROM entry_summaries s \
|
||||
WHERE s.entry_id = e.id AND s.status = 'completed' \
|
||||
AND s.summary_text IS NOT NULL \
|
||||
ORDER BY s.completed_at DESC, s.updated_at DESC, s.id DESC \
|
||||
LIMIT 1\
|
||||
), '')) LIKE ?{n})"
|
||||
OR LOWER(e.entity_kind) LIKE ?{n} OR LOWER(e.visibility) LIKE ?{n})"
|
||||
));
|
||||
params.push(term);
|
||||
}
|
||||
|
|
@ -1511,192 +1483,6 @@ mod tests {
|
|||
assert_eq!(results.len(), 1);
|
||||
}
|
||||
|
||||
fn entry_id_by_title(conn: &rusqlite::Connection, title: &str) -> i64 {
|
||||
conn.query_row(
|
||||
"SELECT id FROM archived_entries WHERE title = ?1",
|
||||
[title],
|
||||
|row| row.get(0),
|
||||
)
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
fn complete_summary(
|
||||
conn: &rusqlite::Connection,
|
||||
entry_id: i64,
|
||||
cache_key: &str,
|
||||
summary_text: &str,
|
||||
) -> String {
|
||||
let uid = database::upsert_pending_entry_summary(
|
||||
conn,
|
||||
entry_id,
|
||||
"test_provider",
|
||||
None,
|
||||
"v1",
|
||||
cache_key,
|
||||
)
|
||||
.unwrap();
|
||||
database::update_entry_summary_status(conn, &uid, "completed", Some(summary_text), None)
|
||||
.unwrap();
|
||||
uid
|
||||
}
|
||||
|
||||
fn set_summary_timestamps(
|
||||
conn: &rusqlite::Connection,
|
||||
summary_uid: &str,
|
||||
timestamp: &str,
|
||||
) {
|
||||
conn.execute(
|
||||
"UPDATE entry_summaries SET completed_at = ?1, updated_at = ?1 WHERE summary_uid = ?2",
|
||||
rusqlite::params![timestamp, summary_uid],
|
||||
)
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn search_summary_json_tags_match_and_unrelated_text_is_absent() {
|
||||
let conn = make_test_db_with_entries();
|
||||
let entry_id = entry_id_by_title(&conn, "Resume Templates");
|
||||
complete_summary(
|
||||
&conn,
|
||||
entry_id,
|
||||
"tags",
|
||||
r#"{"tags":["skincare","dermatology"]}"#,
|
||||
);
|
||||
|
||||
let matches = search_entries(
|
||||
&conn,
|
||||
&SearchEntriesQuery {
|
||||
q: Some("skincare".to_string()),
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(matches.len(), 1);
|
||||
assert_eq!(matches[0].title.as_deref(), Some("Resume Templates"));
|
||||
|
||||
let unrelated = search_entries(
|
||||
&conn,
|
||||
&SearchEntriesQuery {
|
||||
q: Some("neurology".to_string()),
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
assert!(unrelated.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn search_summary_uses_only_the_newest_completed_row() {
|
||||
let conn = make_test_db_with_entries();
|
||||
let entry_id = entry_id_by_title(&conn, "Resume Templates");
|
||||
let older = complete_summary(&conn, entry_id, "older", "legacy-skincare-term");
|
||||
set_summary_timestamps(&conn, &older, "2026-01-01T00:00:00Z");
|
||||
let newer = complete_summary(&conn, entry_id, "newer", "current-dermatology-term");
|
||||
set_summary_timestamps(&conn, &newer, "2026-02-01T00:00:00Z");
|
||||
|
||||
let old_matches = search_entries(
|
||||
&conn,
|
||||
&SearchEntriesQuery {
|
||||
q: Some("legacy-skincare-term".to_string()),
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
assert!(old_matches.is_empty());
|
||||
|
||||
let current_matches = search_entries(
|
||||
&conn,
|
||||
&SearchEntriesQuery {
|
||||
q: Some("current-dermatology-term".to_string()),
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(current_matches.len(), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn search_summary_keeps_latest_completed_when_newer_rows_are_pending_or_failed() {
|
||||
let conn = make_test_db_with_entries();
|
||||
let entry_id = entry_id_by_title(&conn, "Resume Templates");
|
||||
let completed = complete_summary(&conn, entry_id, "completed", "retained-skincare-term");
|
||||
set_summary_timestamps(&conn, &completed, "2026-01-01T00:00:00Z");
|
||||
|
||||
let pending = database::upsert_pending_entry_summary(
|
||||
&conn,
|
||||
entry_id,
|
||||
"test_provider",
|
||||
None,
|
||||
"v1",
|
||||
"pending",
|
||||
)
|
||||
.unwrap();
|
||||
set_summary_timestamps(&conn, &pending, "2026-03-01T00:00:00Z");
|
||||
let failed = database::upsert_pending_entry_summary(
|
||||
&conn,
|
||||
entry_id,
|
||||
"test_provider",
|
||||
None,
|
||||
"v1",
|
||||
"failed",
|
||||
)
|
||||
.unwrap();
|
||||
database::update_entry_summary_status(&conn, &failed, "failed", None, Some("boom")).unwrap();
|
||||
set_summary_timestamps(&conn, &failed, "2026-04-01T00:00:00Z");
|
||||
|
||||
let matches = search_entries(
|
||||
&conn,
|
||||
&SearchEntriesQuery {
|
||||
q: Some("retained-skincare-term".to_string()),
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(matches.len(), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn search_summary_preserves_prefix_collection_and_visibility_scope() {
|
||||
let conn = make_test_db_with_entries();
|
||||
let entry_id = entry_id_by_title(&conn, "Polymarket tweet");
|
||||
complete_summary(&conn, entry_id, "scoped", "scoped-skincare-term");
|
||||
let tag = create_tag(&conn, "/summary-scope").unwrap();
|
||||
database::assign_entry_to_tag(
|
||||
&conn,
|
||||
entry_id,
|
||||
database::get_tag_by_uid(&conn, &tag.tag_uid).unwrap().unwrap().id,
|
||||
)
|
||||
.unwrap();
|
||||
let collection = database::create_collection(&conn, "Summary scope", "summary-scope", 2, false)
|
||||
.unwrap();
|
||||
database::add_entry_to_collection(&conn, collection.id, entry_id, 2).unwrap();
|
||||
|
||||
let query = SearchEntriesQuery {
|
||||
q: Some("scoped-skincare-term".to_string()),
|
||||
source_kind: Some("x".to_string()),
|
||||
entity_kind: Some("tweet".to_string()),
|
||||
url: Some("x.com".to_string()),
|
||||
title: Some("polymarket".to_string()),
|
||||
after: Some("2020-01-01T00:00:00Z".to_string()),
|
||||
before: Some("9999-01-01T00:00:00Z".to_string()),
|
||||
tag: Some("/summary-scope".to_string()),
|
||||
caller_bits: 1,
|
||||
collection_id: Some(collection.id),
|
||||
};
|
||||
assert!(search_entries(&conn, &query).unwrap().is_empty());
|
||||
|
||||
let matches = search_entries(
|
||||
&conn,
|
||||
&SearchEntriesQuery {
|
||||
caller_bits: 2,
|
||||
..query
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(matches.len(), 1);
|
||||
assert_eq!(matches[0].title.as_deref(), Some("Polymarket tweet"));
|
||||
}
|
||||
|
||||
// ---- tag API tests ----
|
||||
|
||||
fn make_tag_test_db() -> (rusqlite::Connection, i64, i64) {
|
||||
|
|
@ -2270,37 +2056,4 @@ mod tests {
|
|||
assert!(guest_children.is_empty(), "guest must not see children of a USER-only collection");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn list_child_entries_orders_by_position_not_archived_at() {
|
||||
let (conn, user_id, run_id) = make_tag_test_db();
|
||||
let container = make_entry_in_db(&conn, user_id, run_id, None, None,
|
||||
"Playlist", "https://example.com/pl");
|
||||
let mk = |title: &str, url: &str| make_entry_in_db(&conn, user_id, run_id,
|
||||
Some(container.id), Some(container.id), title, url);
|
||||
let v1 = mk("V1", "https://example.com/pl/v1");
|
||||
let v2 = mk("V2", "https://example.com/pl/v2");
|
||||
let v3 = mk("V3", "https://example.com/pl/v3");
|
||||
conn.execute(
|
||||
"UPDATE archived_entries SET archived_at = '2000-01-01T00:00:00Z' WHERE id = ?1",
|
||||
[v3.id],
|
||||
).unwrap();
|
||||
let uids = || -> Vec<String> {
|
||||
list_child_entries(&conn, &container.entry_uid, 12).unwrap()
|
||||
.into_iter().map(|e| e.entry_uid).collect()
|
||||
};
|
||||
assert_eq!(uids(), vec![v1.entry_uid.clone(), v2.entry_uid.clone(), v3.entry_uid.clone()]);
|
||||
|
||||
let order = vec![v3.entry_uid.clone(), v1.entry_uid.clone(), v2.entry_uid.clone()];
|
||||
assert_eq!(
|
||||
database::reorder_child_entries(&conn, &container.entry_uid, &order).unwrap(),
|
||||
database::ReorderChildrenOutcome::Reordered
|
||||
);
|
||||
assert_eq!(uids(), order);
|
||||
|
||||
let v4 = mk("V4", "https://example.com/pl/v4");
|
||||
let mut expected = order.clone();
|
||||
expected.push(v4.entry_uid.clone());
|
||||
assert_eq!(uids(), expected);
|
||||
}
|
||||
|
||||
}
|
||||
|
|
|
|||
File diff suppressed because it is too large
Load diff
File diff suppressed because it is too large
Load diff
|
|
@ -1,328 +0,0 @@
|
|||
//! Deno half of the shared yt-dlp tools update (CLI `archivr yt-dlp update` and the admin
|
||||
//! UI): installs the latest official Deno release into
|
||||
//! `<state_dir>/deno/deno`, the JS runtime yt-dlp uses to solve YouTube's challenges.
|
||||
//!
|
||||
//! The flow mirrors the yt-dlp zipapp install: download, extract into a staging file,
|
||||
//! verify it runs (`--version` must report exactly the release version), then rename it
|
||||
//! over the target so a concurrently-running archivr never sees a half-written binary.
|
||||
//! The release `.sha256sum` is not checked: it comes from the same TLS origin as the zip,
|
||||
//! and the zip's CRC32 already catches corruption.
|
||||
|
||||
use anyhow::{bail, Context, Result};
|
||||
use super::js_runtime::{
|
||||
parse_deno_version_output, pinned_deno, probe_deno_version, state_dir_deno, DenoVersion,
|
||||
MIN_DENO_VERSION,
|
||||
};
|
||||
use std::{
|
||||
env, fs,
|
||||
io::{self, Cursor},
|
||||
path::Path,
|
||||
process::Command,
|
||||
time::Duration,
|
||||
};
|
||||
|
||||
/// GitHub release metadata endpoint for the upstream Deno project.
|
||||
pub const DENO_LATEST_RELEASE: &str =
|
||||
"https://api.github.com/repos/denoland/deno/releases/latest";
|
||||
|
||||
/// The Deno zip is ~40 MB; reqwest's blocking client defaults to a 30s total timeout,
|
||||
/// which is too short on slow links. Applies to the zip download only.
|
||||
const DENO_DOWNLOAD_TIMEOUT: Duration = Duration::from_secs(600);
|
||||
|
||||
/// Why a staged prebuilt Deno can fail to spawn even though the file exists: the official
|
||||
/// binaries are dynamically linked against a glibc loader NixOS doesn't provide.
|
||||
const NO_LOADER: &str = "prebuilt deno cannot execute on this host \
|
||||
(missing dynamic loader — on NixOS enable programs.nix-ld)";
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct DenoRelease {
|
||||
pub tag: String,
|
||||
pub version: DenoVersion,
|
||||
pub download_url: String,
|
||||
}
|
||||
|
||||
/// Official release asset for a `std::env::consts::{OS, ARCH}` pair.
|
||||
pub fn deno_release_asset(os: &str, arch: &str) -> Option<&'static str> {
|
||||
match (os, arch) {
|
||||
("macos", "aarch64") => Some("deno-aarch64-apple-darwin.zip"),
|
||||
("macos", "x86_64") => Some("deno-x86_64-apple-darwin.zip"),
|
||||
("linux", "x86_64") => Some("deno-x86_64-unknown-linux-gnu.zip"),
|
||||
("linux", "aarch64") => Some("deno-aarch64-unknown-linux-gnu.zip"),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
/// Extracts the version and download URL from a GitHub "latest release" response.
|
||||
/// The URL is built from the tag and asset name rather than taken from the response.
|
||||
pub fn parse_deno_release(json: &serde_json::Value, asset: &str) -> Result<DenoRelease> {
|
||||
let tag = json
|
||||
.get("tag_name")
|
||||
.and_then(serde_json::Value::as_str)
|
||||
.context("GitHub releases API response had no tag_name")?;
|
||||
// The tag ends up in a URL path; only accept plain version-ish characters.
|
||||
if tag.is_empty()
|
||||
|| !tag
|
||||
.chars()
|
||||
.all(|c| c.is_ascii_alphanumeric() || matches!(c, '.' | '-' | '+'))
|
||||
{
|
||||
bail!("unexpected deno release tag {tag:?}");
|
||||
}
|
||||
let version = DenoVersion::parse(tag)
|
||||
.with_context(|| format!("could not parse a version from deno release tag {tag:?}"))?;
|
||||
let has_asset = json
|
||||
.get("assets")
|
||||
.and_then(serde_json::Value::as_array)
|
||||
.is_some_and(|assets| {
|
||||
assets
|
||||
.iter()
|
||||
.any(|a| a.get("name").and_then(serde_json::Value::as_str) == Some(asset))
|
||||
});
|
||||
if !has_asset {
|
||||
bail!("deno release {tag} has no {asset} asset");
|
||||
}
|
||||
Ok(DenoRelease {
|
||||
tag: tag.to_string(),
|
||||
version,
|
||||
download_url: format!(
|
||||
"https://github.com/denoland/deno/releases/download/{tag}/{asset}"
|
||||
),
|
||||
})
|
||||
}
|
||||
|
||||
/// Installs or updates Deno in the state dir. `Ok` carries a one-line human outcome.
|
||||
pub fn install_deno(client: &reqwest::blocking::Client, log: &mut dyn FnMut(&str)) -> Result<String> {
|
||||
let target = state_dir_deno().context("could not determine a state directory (is $HOME set?)")?;
|
||||
let dir = target
|
||||
.parent()
|
||||
.context("state-dir deno path has no parent directory")?;
|
||||
let staging = dir.join("deno.new");
|
||||
|
||||
let (os, arch) = (env::consts::OS, env::consts::ARCH);
|
||||
let asset = deno_release_asset(os, arch)
|
||||
.with_context(|| format!("unsupported platform {os}/{arch}"))?;
|
||||
|
||||
let body = client
|
||||
.get(DENO_LATEST_RELEASE)
|
||||
.send()
|
||||
.context("failed to reach the GitHub releases API")?
|
||||
.error_for_status()
|
||||
.context("GitHub releases API returned an error")?
|
||||
.text()
|
||||
.context("failed to read the GitHub releases API response")?;
|
||||
let json: serde_json::Value =
|
||||
serde_json::from_str(&body).context("GitHub releases API returned invalid JSON")?;
|
||||
let release = parse_deno_release(&json, asset)?;
|
||||
|
||||
if let Some(installed) = probe_deno_version(&target).filter(|v| *v >= release.version) {
|
||||
return Ok(format!("deno {installed} already installed at {}", target.display()));
|
||||
}
|
||||
|
||||
log(&format!("Downloading deno {}…", release.version));
|
||||
let bytes = client
|
||||
.get(&release.download_url)
|
||||
.timeout(DENO_DOWNLOAD_TIMEOUT)
|
||||
.send()
|
||||
.with_context(|| format!("failed to download {}", release.download_url))?
|
||||
.error_for_status()
|
||||
.with_context(|| format!("download of {} failed", release.download_url))?
|
||||
.bytes()
|
||||
.context("failed to read the downloaded deno zip")?;
|
||||
|
||||
fs::create_dir_all(dir).with_context(|| format!("failed to create {}", dir.display()))?;
|
||||
|
||||
let staged = extract_deno(&bytes, &staging)
|
||||
.and_then(|()| verify_staged(&staging, release.version))
|
||||
.and_then(|runs| {
|
||||
if runs {
|
||||
fs::rename(&staging, &target)
|
||||
.with_context(|| format!("failed to install {}", target.display()))?;
|
||||
}
|
||||
Ok(runs)
|
||||
});
|
||||
let runs = staged.inspect_err(|_| {
|
||||
let _ = fs::remove_file(&staging);
|
||||
})?;
|
||||
|
||||
if !runs {
|
||||
let _ = fs::remove_file(&staging);
|
||||
let pinned_usable = pinned_deno()
|
||||
.and_then(|p| probe_deno_version(&p))
|
||||
.is_some_and(|v| v >= MIN_DENO_VERSION);
|
||||
if pinned_usable {
|
||||
return Ok(format!("skipped: {NO_LOADER}; using pinned ARCHIVR_DENO"));
|
||||
}
|
||||
bail!("{NO_LOADER}, and no usable pinned ARCHIVR_DENO is set");
|
||||
}
|
||||
|
||||
Ok(format!("installed deno {} to {}", release.version, target.display()))
|
||||
}
|
||||
|
||||
/// Writes the zip's `deno` entry to `staging` and makes it executable.
|
||||
fn extract_deno(zip_bytes: &[u8], staging: &Path) -> Result<()> {
|
||||
let mut archive =
|
||||
zip::ZipArchive::new(Cursor::new(zip_bytes)).context("downloaded deno zip is invalid")?;
|
||||
let mut entry = archive
|
||||
.by_name("deno")
|
||||
.context("downloaded deno zip has no `deno` entry")?;
|
||||
let mut out = fs::File::create(staging)
|
||||
.with_context(|| format!("failed to create {}", staging.display()))?;
|
||||
io::copy(&mut entry, &mut out)
|
||||
.with_context(|| format!("failed to extract deno to {}", staging.display()))?;
|
||||
out.sync_all()
|
||||
.with_context(|| format!("failed to flush {}", staging.display()))?;
|
||||
|
||||
#[cfg(unix)]
|
||||
{
|
||||
use std::os::unix::fs::PermissionsExt;
|
||||
fs::set_permissions(staging, fs::Permissions::from_mode(0o755))
|
||||
.with_context(|| format!("failed to chmod +x {}", staging.display()))?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Runs `staging --version` and requires it to report exactly `expected`.
|
||||
///
|
||||
/// `Ok(false)` means the binary exists but the OS could not execute it at all
|
||||
/// (spawn failed with `NotFound` — e.g. the ELF interpreter is missing on NixOS);
|
||||
/// any other failure is an error.
|
||||
fn verify_staged(staging: &Path, expected: DenoVersion) -> Result<bool> {
|
||||
let output = match Command::new(staging).arg("--version").output() {
|
||||
Ok(output) => output,
|
||||
Err(e) if e.kind() == io::ErrorKind::NotFound && staging.is_file() => return Ok(false),
|
||||
Err(e) => {
|
||||
return Err(e).with_context(|| format!("failed to run {} --version", staging.display()));
|
||||
}
|
||||
};
|
||||
let got = output
|
||||
.status
|
||||
.success()
|
||||
.then(|| parse_deno_version_output(&String::from_utf8_lossy(&output.stdout)))
|
||||
.flatten();
|
||||
if got != Some(expected) {
|
||||
let got = got.map_or_else(|| "no parseable version".to_string(), |v| v.to_string());
|
||||
bail!("downloaded deno failed verification (expected {expected}, got {got})");
|
||||
}
|
||||
Ok(true)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::{deno_release_asset, parse_deno_release, verify_staged};
|
||||
#[cfg(unix)]
|
||||
use crate::downloader::write_script;
|
||||
use crate::downloader::js_runtime::DenoVersion;
|
||||
use serde_json::json;
|
||||
|
||||
#[test]
|
||||
fn supported_platforms_map_to_assets() {
|
||||
assert_eq!(
|
||||
deno_release_asset("macos", "aarch64"),
|
||||
Some("deno-aarch64-apple-darwin.zip")
|
||||
);
|
||||
assert_eq!(
|
||||
deno_release_asset("macos", "x86_64"),
|
||||
Some("deno-x86_64-apple-darwin.zip")
|
||||
);
|
||||
assert_eq!(
|
||||
deno_release_asset("linux", "x86_64"),
|
||||
Some("deno-x86_64-unknown-linux-gnu.zip")
|
||||
);
|
||||
assert_eq!(
|
||||
deno_release_asset("linux", "aarch64"),
|
||||
Some("deno-aarch64-unknown-linux-gnu.zip")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn unsupported_platforms_have_no_asset() {
|
||||
assert_eq!(deno_release_asset("windows", "x86_64"), None);
|
||||
assert_eq!(deno_release_asset("linux", "riscv64"), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn release_json_yields_version_and_url() {
|
||||
let json = json!({
|
||||
"tag_name": "v2.9.7",
|
||||
"assets": [
|
||||
{"name": "deno-x86_64-unknown-linux-gnu.zip"},
|
||||
{"name": "deno-aarch64-apple-darwin.zip"},
|
||||
],
|
||||
});
|
||||
let release = parse_deno_release(&json, "deno-aarch64-apple-darwin.zip").unwrap();
|
||||
assert_eq!(release.tag, "v2.9.7");
|
||||
assert_eq!(
|
||||
release.version,
|
||||
DenoVersion { major: 2, minor: 9, patch: 7 }
|
||||
);
|
||||
assert_eq!(
|
||||
release.download_url,
|
||||
"https://github.com/denoland/deno/releases/download/v2.9.7/deno-aarch64-apple-darwin.zip"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn missing_asset_is_an_error_naming_it() {
|
||||
let json = json!({
|
||||
"tag_name": "v2.9.7",
|
||||
"assets": [{"name": "deno-x86_64-unknown-linux-gnu.zip"}],
|
||||
});
|
||||
let err = parse_deno_release(&json, "deno-aarch64-apple-darwin.zip").unwrap_err();
|
||||
assert!(
|
||||
err.to_string().contains("deno-aarch64-apple-darwin.zip"),
|
||||
"{err:#}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn missing_or_bad_tag_is_an_error() {
|
||||
let assets = json!([{"name": "deno-aarch64-apple-darwin.zip"}]);
|
||||
for json in [
|
||||
json!({"assets": assets}),
|
||||
json!({"tag_name": 297, "assets": assets}),
|
||||
json!({"tag_name": "nightly", "assets": assets}),
|
||||
json!({"tag_name": "v2.9.7/../../evil", "assets": assets}),
|
||||
] {
|
||||
assert!(
|
||||
parse_deno_release(&json, "deno-aarch64-apple-darwin.zip").is_err(),
|
||||
"{json}"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
#[test]
|
||||
fn staged_binary_must_report_the_release_version() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let expected = DenoVersion { major: 2, minor: 9, patch: 7 };
|
||||
|
||||
// A fresh path per script: rewriting one path that was just exec'd invites ETXTBSY.
|
||||
let good = tmp.path().join("good/deno.new");
|
||||
write_script(&good, "#!/bin/sh\necho 'deno 2.9.7 (stable, release, test)'\n");
|
||||
assert!(verify_staged(&good, expected).unwrap());
|
||||
|
||||
let wrong = tmp.path().join("wrong/deno.new");
|
||||
write_script(&wrong, "#!/bin/sh\necho 'deno 2.9.6 (stable, release, test)'\n");
|
||||
let err = verify_staged(&wrong, expected).unwrap_err();
|
||||
assert!(err.to_string().contains("expected 2.9.7, got 2.9.6"), "{err:#}");
|
||||
|
||||
let failing = tmp.path().join("failing/deno.new");
|
||||
write_script(&failing, "#!/bin/sh\nexit 1\n");
|
||||
assert!(verify_staged(&failing, expected).is_err());
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
#[test]
|
||||
fn existing_but_unexecutable_binary_is_reported_as_cannot_run() {
|
||||
// A script whose interpreter is missing fails to spawn with NotFound even though
|
||||
// the file exists — the same shape as a glibc ELF on NixOS without nix-ld.
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let staged = tmp.path().join("deno.new");
|
||||
write_script(&staged, "#!/nonexistent/ld-linux.so\n");
|
||||
let expected = DenoVersion { major: 2, minor: 9, patch: 7 };
|
||||
assert!(!verify_staged(&staged, expected).unwrap());
|
||||
|
||||
// A genuinely missing file is an error, not "cannot execute".
|
||||
let missing = tmp.path().join("missing");
|
||||
assert!(verify_staged(&missing, expected).is_err());
|
||||
}
|
||||
}
|
||||
|
|
@ -1,684 +0,0 @@
|
|||
//! JavaScript runtime resolution for yt-dlp.
|
||||
//!
|
||||
//! yt-dlp needs a JS runtime to solve YouTube's challenges (EJS). Without one, YouTube
|
||||
//! downloads may fail with HTTP 403. This module picks the runtime archivr passes to every
|
||||
//! yt-dlp process via `--js-runtimes`:
|
||||
//!
|
||||
//! 1. `ARCHIVR_JS_RUNTIME` (forced, `RUNTIME[:ABS_PATH]`; skips resolution and version checks).
|
||||
//! 2. The newest Deno >= [`MIN_DENO_VERSION`] among the pinned `ARCHIVR_DENO` and the
|
||||
//! state-dir copy (`<state_dir>/deno/deno`); an exact tie goes to the state dir.
|
||||
//! 3. `deno` on `PATH`, if new enough.
|
||||
//! 4. Nothing (a warning is printed once per resolution: first use and each refresh).
|
||||
//!
|
||||
//! Only Deno is ever chosen automatically; Node, Bun and QuickJS are used only when forced.
|
||||
|
||||
use std::{
|
||||
env,
|
||||
ffi::{OsStr, OsString},
|
||||
fmt,
|
||||
path::{Path, PathBuf},
|
||||
process::Command,
|
||||
sync::RwLock,
|
||||
};
|
||||
|
||||
use super::ytdlp::state_dir;
|
||||
|
||||
/// Forced runtime override, `RUNTIME[:ABS_PATH]` with RUNTIME one of deno|node|bun|quickjs.
|
||||
pub const JS_RUNTIME_ENV: &str = "ARCHIVR_JS_RUNTIME";
|
||||
/// Pinned Deno binary (set by the Nix wrappers and the Docker image).
|
||||
pub const DENO_ENV: &str = "ARCHIVR_DENO";
|
||||
/// Oldest Deno yt-dlp's EJS solver supports.
|
||||
pub const MIN_DENO_VERSION: DenoVersion = DenoVersion { major: 2, minor: 3, patch: 0 };
|
||||
|
||||
/// Cached choice: outer `None` = not resolved yet. Swapped by [`refresh_js_runtime`].
|
||||
static RESOLVED_JS_RUNTIME: RwLock<Option<Option<JsRuntime>>> = RwLock::new(None);
|
||||
|
||||
/// Resolves (uncached) and prints the warnings; runs on first use and on each refresh.
|
||||
fn resolve_js_runtime_logged() -> Option<JsRuntime> {
|
||||
if let Err(reason) = forced_js_runtime() {
|
||||
let raw = env::var_os(JS_RUNTIME_ENV).unwrap_or_default();
|
||||
eprintln!("warn: ignoring {JS_RUNTIME_ENV}={raw:?}: {reason}");
|
||||
}
|
||||
let resolved = resolve_js_runtime_with_role().map(|(_, rt)| rt);
|
||||
if resolved.is_none() {
|
||||
eprintln!(
|
||||
"warn: no JavaScript runtime for yt-dlp (need deno >= {MIN_DENO_VERSION} via \
|
||||
{DENO_ENV}, the state dir, or PATH; or set {JS_RUNTIME_ENV}) — YouTube \
|
||||
downloads may fail with HTTP 403; run `archivr yt-dlp update` to install deno"
|
||||
);
|
||||
}
|
||||
resolved
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum JsRuntimeKind {
|
||||
Deno,
|
||||
Node,
|
||||
Bun,
|
||||
QuickJs,
|
||||
}
|
||||
|
||||
impl JsRuntimeKind {
|
||||
/// Runtime name as yt-dlp's `--js-runtimes` expects it.
|
||||
pub fn as_str(self) -> &'static str {
|
||||
match self {
|
||||
Self::Deno => "deno",
|
||||
Self::Node => "node",
|
||||
Self::Bun => "bun",
|
||||
Self::QuickJs => "quickjs",
|
||||
}
|
||||
}
|
||||
|
||||
/// ASCII case-insensitive match against the exact allowlist.
|
||||
pub fn from_name(name: &str) -> Option<Self> {
|
||||
[Self::Deno, Self::Node, Self::Bun, Self::QuickJs]
|
||||
.into_iter()
|
||||
.find(|k| k.as_str().eq_ignore_ascii_case(name))
|
||||
}
|
||||
|
||||
/// Executable name yt-dlp looks for when the runtime path is a directory
|
||||
/// (mirrors `_determine_runtime_path` in yt-dlp's JS runtime classes).
|
||||
fn executable_name(self) -> &'static str {
|
||||
match self {
|
||||
Self::QuickJs => "qjs",
|
||||
other => other.as_str(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct JsRuntime {
|
||||
pub kind: JsRuntimeKind,
|
||||
pub path: Option<PathBuf>,
|
||||
}
|
||||
|
||||
impl JsRuntime {
|
||||
/// `kind` or `kind:path`, built without lossy conversion.
|
||||
pub fn spec(&self) -> OsString {
|
||||
let mut spec = OsString::from(self.kind.as_str());
|
||||
if let Some(path) = &self.path {
|
||||
spec.push(":");
|
||||
spec.push(path.as_os_str());
|
||||
}
|
||||
spec
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)]
|
||||
pub struct DenoVersion {
|
||||
pub major: u64,
|
||||
pub minor: u64,
|
||||
pub patch: u64,
|
||||
}
|
||||
|
||||
impl DenoVersion {
|
||||
/// Parses `2.9.7` or `v2.9.7`; anything after the patch digits (`+abc`, `-rc1`) is ignored.
|
||||
pub fn parse(s: &str) -> Option<Self> {
|
||||
let s = s.trim();
|
||||
let s = s.strip_prefix('v').unwrap_or(s);
|
||||
let end = s
|
||||
.find(|c: char| !(c.is_ascii_digit() || c == '.'))
|
||||
.unwrap_or(s.len());
|
||||
let mut parts = s[..end].split('.').map(|p| p.parse::<u64>().ok());
|
||||
let version = Self {
|
||||
major: parts.next()??,
|
||||
minor: parts.next()??,
|
||||
patch: parts.next()??,
|
||||
};
|
||||
parts.next().is_none().then_some(version)
|
||||
}
|
||||
}
|
||||
|
||||
impl fmt::Display for DenoVersion {
|
||||
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
|
||||
write!(f, "{}.{}.{}", self.major, self.minor, self.patch)
|
||||
}
|
||||
}
|
||||
|
||||
/// Validates a `RUNTIME[:ABS_PATH]` spec. The path may be a file or a directory
|
||||
/// (yt-dlp accepts both) but must be absolute and exist.
|
||||
pub fn parse_js_runtime_spec(raw: &str) -> Result<JsRuntime, String> {
|
||||
let raw = raw.trim();
|
||||
let (name, path) = match raw.split_once(':') {
|
||||
Some((name, path)) => (name, Some(path)),
|
||||
None => (raw, None),
|
||||
};
|
||||
let kind = JsRuntimeKind::from_name(name).ok_or_else(|| {
|
||||
format!("unknown runtime {name} (expected deno, node, bun or quickjs)")
|
||||
})?;
|
||||
let path = match path {
|
||||
None => None,
|
||||
Some("") => return Err("empty path after ':'".to_string()),
|
||||
Some(p) => {
|
||||
let p = PathBuf::from(p);
|
||||
if !p.is_absolute() {
|
||||
return Err("path must be absolute".to_string());
|
||||
}
|
||||
if !p.exists() {
|
||||
return Err("path does not exist".to_string());
|
||||
}
|
||||
Some(p)
|
||||
}
|
||||
};
|
||||
Ok(JsRuntime { kind, path })
|
||||
}
|
||||
|
||||
/// Reads `ARCHIVR_JS_RUNTIME`; `Ok(None)` if unset or empty.
|
||||
pub fn forced_js_runtime() -> Result<Option<JsRuntime>, String> {
|
||||
match env::var(JS_RUNTIME_ENV) {
|
||||
Err(env::VarError::NotPresent) => Ok(None),
|
||||
Err(env::VarError::NotUnicode(_)) => Err("value is not valid UTF-8".to_string()),
|
||||
Ok(raw) if raw.trim().is_empty() => Ok(None),
|
||||
Ok(raw) => parse_js_runtime_spec(&raw).map(Some),
|
||||
}
|
||||
}
|
||||
|
||||
/// `ARCHIVR_DENO`, if it points to an existing file.
|
||||
pub fn pinned_deno() -> Option<PathBuf> {
|
||||
env::var_os(DENO_ENV)
|
||||
.filter(|v| !v.is_empty())
|
||||
.map(PathBuf::from)
|
||||
.filter(|p| p.is_file())
|
||||
}
|
||||
|
||||
/// Deno slot inside the state dir (`<state_dir>/deno/deno`); not existence-filtered.
|
||||
pub fn state_dir_deno() -> Option<PathBuf> {
|
||||
state_dir().map(|d| d.join("deno").join("deno"))
|
||||
}
|
||||
|
||||
/// First `dir/name` that is a file, scanning `path_var` like a shell would.
|
||||
pub fn find_on_path(name: &str, path_var: Option<&OsStr>) -> Option<PathBuf> {
|
||||
env::split_paths(path_var?)
|
||||
.filter(|dir| !dir.as_os_str().is_empty())
|
||||
.map(|dir| dir.join(name))
|
||||
.find(|candidate| candidate.is_file())
|
||||
}
|
||||
|
||||
/// `deno` on the process `PATH`.
|
||||
pub fn path_deno() -> Option<PathBuf> {
|
||||
find_on_path("deno", env::var_os("PATH").as_deref())
|
||||
}
|
||||
|
||||
/// Parses `deno --version` output (`deno 2.9.7 (stable, release, ...)` on the first line).
|
||||
pub fn parse_deno_version_output(stdout: &str) -> Option<DenoVersion> {
|
||||
let first = stdout.lines().next()?.trim();
|
||||
let rest = first.strip_prefix("deno ")?;
|
||||
DenoVersion::parse(rest.split_whitespace().next()?)
|
||||
}
|
||||
|
||||
/// Runs `<binary> --version` and parses it; `None` on spawn failure, non-zero exit or junk output.
|
||||
pub fn probe_deno_version(binary: &Path) -> Option<DenoVersion> {
|
||||
let output = Command::new(binary).arg("--version").output().ok()?;
|
||||
if !output.status.success() {
|
||||
return None;
|
||||
}
|
||||
parse_deno_version_output(&String::from_utf8_lossy(&output.stdout))
|
||||
}
|
||||
|
||||
/// Human version string for a (typically forced) runtime. Deno is probed and parsed; other
|
||||
/// runtimes with a path report the first stdout line of `--version`; pathless non-Deno → `None`.
|
||||
/// A directory path gets the runtime's executable name joined, as yt-dlp does.
|
||||
pub fn probe_js_runtime_version(rt: &JsRuntime) -> Option<String> {
|
||||
let binary = match &rt.path {
|
||||
Some(p) if p.is_dir() => p.join(rt.kind.executable_name()),
|
||||
Some(p) => p.clone(),
|
||||
None if rt.kind == JsRuntimeKind::Deno => PathBuf::from("deno"),
|
||||
None => return None,
|
||||
};
|
||||
if rt.kind == JsRuntimeKind::Deno {
|
||||
return probe_deno_version(&binary).map(|v| v.to_string());
|
||||
}
|
||||
let output = Command::new(&binary).arg("--version").output().ok()?;
|
||||
if !output.status.success() {
|
||||
return None;
|
||||
}
|
||||
String::from_utf8_lossy(&output.stdout)
|
||||
.lines()
|
||||
.next()
|
||||
.map(|l| l.trim().to_string())
|
||||
.filter(|l| !l.is_empty())
|
||||
}
|
||||
|
||||
/// Which candidate slot a resolved runtime came from. Several slots can point at the same
|
||||
/// binary (the Nix wrappers set `ARCHIVR_DENO` and also put that Deno on `PATH`), so callers
|
||||
/// that need to name the winner must use the role, not compare paths.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum JsRuntimeRole {
|
||||
/// `ARCHIVR_JS_RUNTIME`.
|
||||
Forced,
|
||||
/// `ARCHIVR_DENO`.
|
||||
Pinned,
|
||||
/// `<state_dir>/deno/deno`.
|
||||
StateDir,
|
||||
/// `deno` on `PATH`.
|
||||
Path,
|
||||
}
|
||||
|
||||
impl JsRuntimeRole {
|
||||
/// Row label used by `archivr yt-dlp status`.
|
||||
pub fn label(self) -> &'static str {
|
||||
match self {
|
||||
Self::Forced => "force (ARCHIVR_JS_RUNTIME)",
|
||||
Self::Pinned => "env (ARCHIVR_DENO)",
|
||||
Self::StateDir => "state-dir",
|
||||
Self::Path => "path (deno)",
|
||||
}
|
||||
}
|
||||
|
||||
/// Stable machine key (API `role`): "force" | "env" | "state-dir" | "path".
|
||||
pub fn key(self) -> &'static str {
|
||||
match self {
|
||||
Self::Forced => "force",
|
||||
Self::Pinned => "env",
|
||||
Self::StateDir => "state-dir",
|
||||
Self::Path => "path",
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Existing Deno candidates, pinned first and state-dir last (so ties go to the state dir).
|
||||
pub fn deno_candidates() -> Vec<(JsRuntimeRole, PathBuf)> {
|
||||
[
|
||||
(JsRuntimeRole::Pinned, pinned_deno()),
|
||||
(JsRuntimeRole::StateDir, state_dir_deno()),
|
||||
]
|
||||
.into_iter()
|
||||
.filter_map(|(role, p)| p.filter(|p| p.is_file()).map(|p| (role, p)))
|
||||
.collect()
|
||||
}
|
||||
|
||||
pub(crate) fn resolve_js_runtime_with_path(
|
||||
path_var: Option<&OsStr>,
|
||||
) -> Option<(JsRuntimeRole, JsRuntime)> {
|
||||
if let Ok(Some(rt)) = forced_js_runtime() {
|
||||
return Some((JsRuntimeRole::Forced, rt));
|
||||
}
|
||||
let usable = |p: &Path| probe_deno_version(p).filter(|v| *v >= MIN_DENO_VERSION);
|
||||
// `max_by` keeps the last maximum, so an exact tie goes to the state dir.
|
||||
let (role, best) = deno_candidates()
|
||||
.into_iter()
|
||||
.filter_map(|(role, p)| usable(&p).map(|v| (v, role, p)))
|
||||
.max_by(|(a, ..), (b, ..)| a.cmp(b))
|
||||
.map(|(_, role, p)| (role, p))
|
||||
.or_else(|| {
|
||||
find_on_path("deno", path_var)
|
||||
.filter(|p| usable(p).is_some())
|
||||
.map(|p| (JsRuntimeRole::Path, p))
|
||||
})?;
|
||||
Some((role, JsRuntime { kind: JsRuntimeKind::Deno, path: Some(best) }))
|
||||
}
|
||||
|
||||
/// Resolves without caching or printing, also reporting which candidate slot won
|
||||
/// (used by `archivr yt-dlp status` to mark exactly one row).
|
||||
pub fn resolve_js_runtime_with_role() -> Option<(JsRuntimeRole, JsRuntime)> {
|
||||
resolve_js_runtime_with_path(env::var_os("PATH").as_deref())
|
||||
}
|
||||
|
||||
/// Cached until [`refresh_js_runtime`]; owned clone so a refresh never invalidates a caller.
|
||||
pub fn resolve_js_runtime() -> Option<JsRuntime> {
|
||||
if let Some(v) = RESOLVED_JS_RUNTIME.read().unwrap_or_else(|e| e.into_inner()).as_ref() {
|
||||
return v.clone();
|
||||
}
|
||||
let mut slot = RESOLVED_JS_RUNTIME.write().unwrap_or_else(|e| e.into_inner());
|
||||
slot.get_or_insert_with(resolve_js_runtime_logged).clone()
|
||||
}
|
||||
|
||||
/// Re-resolves (outside the lock) and swaps the cache; called after a successful Deno
|
||||
/// install. Commands already built keep their old runtime.
|
||||
pub fn refresh_js_runtime() -> Option<JsRuntime> {
|
||||
let fresh = resolve_js_runtime_logged();
|
||||
*RESOLVED_JS_RUNTIME.write().unwrap_or_else(|e| e.into_inner()) = Some(fresh.clone());
|
||||
fresh
|
||||
}
|
||||
|
||||
/// yt-dlp arguments selecting `runtime`.
|
||||
///
|
||||
/// yt-dlp builds its runtime map keyed by name, splitting each `--js-runtimes` value on the
|
||||
/// first `:` (`yt_dlp/__init__.py:784-786`), so a later entry for the same name wins:
|
||||
/// `deno:<path>` replaces the default `deno` entry. Non-Deno runtimes are preceded by
|
||||
/// `--no-js-runtimes` (`options.py:460-479`) so a Deno yt-dlp finds on its own can't take
|
||||
/// priority over the forced choice. Each value is one argv element; no shell is involved.
|
||||
pub fn js_runtime_args(runtime: Option<&JsRuntime>) -> Vec<OsString> {
|
||||
let Some(rt) = runtime else {
|
||||
return Vec::new();
|
||||
};
|
||||
let mut args = Vec::with_capacity(3);
|
||||
if rt.kind != JsRuntimeKind::Deno {
|
||||
args.push(OsString::from("--no-js-runtimes"));
|
||||
}
|
||||
args.push(OsString::from("--js-runtimes"));
|
||||
args.push(rt.spec());
|
||||
args
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::downloader::ytdlp::STATE_DIR_ENV;
|
||||
use std::{fs, sync::MutexGuard};
|
||||
use tempfile::TempDir;
|
||||
|
||||
/// Serialises env-mutating tests and clears every var the resolver reads.
|
||||
fn env_guard() -> MutexGuard<'static, ()> {
|
||||
let guard = crate::downloader::RESOLVER_ENV_LOCK
|
||||
.lock()
|
||||
.unwrap_or_else(|e| e.into_inner());
|
||||
for key in [JS_RUNTIME_ENV, DENO_ENV, STATE_DIR_ENV] {
|
||||
unsafe { env::remove_var(key) };
|
||||
}
|
||||
guard
|
||||
}
|
||||
|
||||
/// Points the state dir at an (initially empty) dir under `tmp` and returns it.
|
||||
fn set_state(tmp: &TempDir) -> PathBuf {
|
||||
let state = tmp.path().join("state");
|
||||
fs::create_dir_all(&state).unwrap();
|
||||
unsafe { env::set_var(STATE_DIR_ENV, &state) };
|
||||
state
|
||||
}
|
||||
|
||||
/// Writes a fake `deno` script to `path` (every caller uses a fresh path), then waits until
|
||||
/// it can be exec'd. A child forked by a parallel test while our write fd was open keeps a
|
||||
/// copy of it until that child execs, so our exec can fail with ETXTBSY
|
||||
/// (rust-lang/rust#114554). One successful exec proves no writer is left, and none can
|
||||
/// appear later because our fd is already closed.
|
||||
fn fake_deno(path: &Path, ver: &str) {
|
||||
use std::os::unix::fs::PermissionsExt;
|
||||
fs::create_dir_all(path.parent().unwrap()).unwrap();
|
||||
fs::write(
|
||||
path,
|
||||
format!("#!/bin/sh\necho 'deno {ver} (stable, release, test)'\necho 'v8 1.0'\n"),
|
||||
)
|
||||
.unwrap();
|
||||
fs::set_permissions(path, fs::Permissions::from_mode(0o755)).unwrap();
|
||||
for _ in 0..200 {
|
||||
match Command::new(path).arg("--version").output() {
|
||||
Err(e) if e.kind() == std::io::ErrorKind::ExecutableFileBusy => {
|
||||
std::thread::sleep(std::time::Duration::from_millis(5));
|
||||
}
|
||||
_ => return,
|
||||
}
|
||||
}
|
||||
panic!("{} stayed busy (ETXTBSY)", path.display());
|
||||
}
|
||||
|
||||
fn deno_at(role: JsRuntimeRole, p: PathBuf) -> Option<(JsRuntimeRole, JsRuntime)> {
|
||||
Some((role, JsRuntime { kind: JsRuntimeKind::Deno, path: Some(p) }))
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn spec_parsing_accepts_allowlisted_runtimes() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let file = tmp.path().join("node");
|
||||
fs::write(&file, "").unwrap();
|
||||
|
||||
assert_eq!(
|
||||
parse_js_runtime_spec("deno"),
|
||||
Ok(JsRuntime { kind: JsRuntimeKind::Deno, path: None })
|
||||
);
|
||||
assert_eq!(parse_js_runtime_spec(" NODE ").unwrap().kind, JsRuntimeKind::Node);
|
||||
assert_eq!(
|
||||
parse_js_runtime_spec(&format!("node:{}", file.display())),
|
||||
Ok(JsRuntime { kind: JsRuntimeKind::Node, path: Some(file.clone()) })
|
||||
);
|
||||
assert_eq!(
|
||||
parse_js_runtime_spec(&format!("deno:{}", tmp.path().display())).unwrap().path,
|
||||
Some(tmp.path().to_path_buf())
|
||||
);
|
||||
assert_eq!(parse_js_runtime_spec("bun").unwrap().kind, JsRuntimeKind::Bun);
|
||||
assert_eq!(parse_js_runtime_spec("quickjs").unwrap().kind, JsRuntimeKind::QuickJs);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn spec_parsing_rejects_bad_values() {
|
||||
assert_eq!(parse_js_runtime_spec("node:"), Err("empty path after ':'".into()));
|
||||
assert_eq!(parse_js_runtime_spec("node:rel/path"), Err("path must be absolute".into()));
|
||||
assert_eq!(
|
||||
parse_js_runtime_spec("deno:/does/not/exist"),
|
||||
Err("path does not exist".into())
|
||||
);
|
||||
for bad in ["python", "--exec", "deno,node"] {
|
||||
let err = parse_js_runtime_spec(bad).unwrap_err();
|
||||
assert!(err.starts_with("unknown runtime"), "{bad}: {err}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn args_follow_runtime_kind() {
|
||||
assert!(js_runtime_args(None).is_empty());
|
||||
let deno = JsRuntime { kind: JsRuntimeKind::Deno, path: Some("/p".into()) };
|
||||
assert_eq!(js_runtime_args(Some(&deno)), ["--js-runtimes", "deno:/p"]);
|
||||
let node = JsRuntime { kind: JsRuntimeKind::Node, path: Some("/p".into()) };
|
||||
assert_eq!(
|
||||
js_runtime_args(Some(&node)),
|
||||
["--no-js-runtimes", "--js-runtimes", "node:/p"]
|
||||
);
|
||||
let bun = JsRuntime { kind: JsRuntimeKind::Bun, path: None };
|
||||
assert_eq!(js_runtime_args(Some(&bun)), ["--no-js-runtimes", "--js-runtimes", "bun"]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn version_parsing() {
|
||||
let v297 = Some(DenoVersion { major: 2, minor: 9, patch: 7 });
|
||||
assert_eq!(
|
||||
parse_deno_version_output(
|
||||
"deno 2.9.7 (stable, release, aarch64-apple-darwin)\nv8 14.0\ntypescript 5.9\n"
|
||||
),
|
||||
v297
|
||||
);
|
||||
assert_eq!(parse_deno_version_output("deno 2.9.7+abc123 (canary, x)\n"), v297);
|
||||
assert_eq!(parse_deno_version_output("node v22"), None);
|
||||
assert_eq!(parse_deno_version_output(""), None);
|
||||
assert_eq!(parse_deno_version_output("deno"), None);
|
||||
assert_eq!(DenoVersion::parse("v2.9.7"), v297);
|
||||
assert_eq!(DenoVersion::parse("2.9"), None);
|
||||
assert_eq!(DenoVersion::parse("2.9.7.1"), None);
|
||||
assert!(DenoVersion::parse("2.10.0") > v297);
|
||||
assert_eq!(v297.unwrap().to_string(), "2.9.7");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn newer_state_dir_deno_wins() {
|
||||
let _g = env_guard();
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let state = set_state(&tmp);
|
||||
let pinned = tmp.path().join("pin/deno");
|
||||
fake_deno(&pinned, "2.4.0");
|
||||
fake_deno(&state.join("deno/deno"), "2.9.7");
|
||||
unsafe { env::set_var(DENO_ENV, &pinned) };
|
||||
assert_eq!(
|
||||
resolve_js_runtime_with_path(None),
|
||||
deno_at(JsRuntimeRole::StateDir, state.join("deno/deno"))
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn newer_pinned_deno_wins() {
|
||||
let _g = env_guard();
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let state = set_state(&tmp);
|
||||
let pinned = tmp.path().join("pin/deno");
|
||||
fake_deno(&pinned, "2.10.0");
|
||||
fake_deno(&state.join("deno/deno"), "2.9.7");
|
||||
unsafe { env::set_var(DENO_ENV, &pinned) };
|
||||
assert_eq!(
|
||||
resolve_js_runtime_with_path(None),
|
||||
deno_at(JsRuntimeRole::Pinned, pinned)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tie_goes_to_state_dir() {
|
||||
let _g = env_guard();
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let state = set_state(&tmp);
|
||||
let pinned = tmp.path().join("pin/deno");
|
||||
fake_deno(&pinned, "2.9.7");
|
||||
fake_deno(&state.join("deno/deno"), "2.9.7");
|
||||
unsafe { env::set_var(DENO_ENV, &pinned) };
|
||||
assert_eq!(
|
||||
resolve_js_runtime_with_path(None),
|
||||
deno_at(JsRuntimeRole::StateDir, state.join("deno/deno"))
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn too_old_pinned_falls_back_to_path() {
|
||||
let _g = env_guard();
|
||||
let tmp = TempDir::new().unwrap();
|
||||
set_state(&tmp);
|
||||
let pinned = tmp.path().join("pin/deno");
|
||||
fake_deno(&pinned, "2.2.9");
|
||||
let bin = tmp.path().join("bin");
|
||||
fake_deno(&bin.join("deno"), "2.4.0");
|
||||
unsafe { env::set_var(DENO_ENV, &pinned) };
|
||||
assert_eq!(
|
||||
resolve_js_runtime_with_path(Some(bin.as_os_str())),
|
||||
deno_at(JsRuntimeRole::Path, bin.join("deno"))
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn valid_pinned_beats_newer_path_deno() {
|
||||
let _g = env_guard();
|
||||
let tmp = TempDir::new().unwrap();
|
||||
set_state(&tmp);
|
||||
let pinned = tmp.path().join("pin/deno");
|
||||
fake_deno(&pinned, "2.4.0");
|
||||
let bin = tmp.path().join("bin");
|
||||
fake_deno(&bin.join("deno"), "2.9.7");
|
||||
unsafe { env::set_var(DENO_ENV, &pinned) };
|
||||
assert_eq!(
|
||||
resolve_js_runtime_with_path(Some(bin.as_os_str())),
|
||||
deno_at(JsRuntimeRole::Pinned, pinned)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn pinned_deno_also_on_path_wins_as_pinned_only() {
|
||||
// The Nix wrappers set ARCHIVR_DENO and put the same Deno on PATH.
|
||||
let _g = env_guard();
|
||||
let tmp = TempDir::new().unwrap();
|
||||
set_state(&tmp);
|
||||
let bin = tmp.path().join("bin");
|
||||
let deno = bin.join("deno");
|
||||
fake_deno(&deno, "2.9.4");
|
||||
unsafe { env::set_var(DENO_ENV, &deno) };
|
||||
assert_eq!(find_on_path("deno", Some(bin.as_os_str())), Some(deno.clone()));
|
||||
assert_eq!(
|
||||
resolve_js_runtime_with_path(Some(bin.as_os_str())),
|
||||
deno_at(JsRuntimeRole::Pinned, deno)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn forced_pinned_deno_wins_as_forced_only() {
|
||||
let _g = env_guard();
|
||||
let tmp = TempDir::new().unwrap();
|
||||
set_state(&tmp);
|
||||
let bin = tmp.path().join("bin");
|
||||
let deno = bin.join("deno");
|
||||
fake_deno(&deno, "2.9.4");
|
||||
unsafe {
|
||||
env::set_var(DENO_ENV, &deno);
|
||||
env::set_var(JS_RUNTIME_ENV, format!("deno:{}", deno.display()));
|
||||
}
|
||||
assert_eq!(
|
||||
resolve_js_runtime_with_path(Some(bin.as_os_str())),
|
||||
deno_at(JsRuntimeRole::Forced, deno)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn too_old_path_deno_resolves_none() {
|
||||
let _g = env_guard();
|
||||
let tmp = TempDir::new().unwrap();
|
||||
set_state(&tmp);
|
||||
let bin = tmp.path().join("bin");
|
||||
fake_deno(&bin.join("deno"), "2.2.9");
|
||||
assert_eq!(resolve_js_runtime_with_path(Some(bin.as_os_str())), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn valid_forced_runtime_wins() {
|
||||
let _g = env_guard();
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let state = set_state(&tmp);
|
||||
let pinned = tmp.path().join("pin/deno");
|
||||
fake_deno(&pinned, "2.9.7");
|
||||
fake_deno(&state.join("deno/deno"), "2.9.7");
|
||||
let node = tmp.path().join("node");
|
||||
fs::write(&node, "").unwrap();
|
||||
unsafe {
|
||||
env::set_var(DENO_ENV, &pinned);
|
||||
env::set_var(JS_RUNTIME_ENV, format!("node:{}", node.display()));
|
||||
}
|
||||
assert_eq!(
|
||||
resolve_js_runtime_with_path(None),
|
||||
Some((
|
||||
JsRuntimeRole::Forced,
|
||||
JsRuntime { kind: JsRuntimeKind::Node, path: Some(node) }
|
||||
))
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn invalid_forced_runtime_is_ignored() {
|
||||
let _g = env_guard();
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let state = set_state(&tmp);
|
||||
fake_deno(&state.join("deno/deno"), "2.9.7");
|
||||
unsafe { env::set_var(JS_RUNTIME_ENV, "python") };
|
||||
assert!(forced_js_runtime().is_err());
|
||||
assert_eq!(
|
||||
resolve_js_runtime_with_path(None),
|
||||
deno_at(JsRuntimeRole::StateDir, state.join("deno/deno"))
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn no_candidates_resolves_none() {
|
||||
let _g = env_guard();
|
||||
let tmp = TempDir::new().unwrap();
|
||||
set_state(&tmp);
|
||||
let empty_bin = tmp.path().join("bin");
|
||||
fs::create_dir_all(&empty_bin).unwrap();
|
||||
assert!(deno_candidates().is_empty());
|
||||
assert_eq!(resolve_js_runtime_with_path(Some(empty_bin.as_os_str())), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn forced_directory_probe_joins_runtime_name() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
fake_deno(&tmp.path().join("deno"), "2.9.7");
|
||||
let rt = JsRuntime { kind: JsRuntimeKind::Deno, path: Some(tmp.path().to_path_buf()) };
|
||||
assert_eq!(probe_js_runtime_version(&rt).as_deref(), Some("2.9.7"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn find_on_path_scans_in_order() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let (a, b) = (tmp.path().join("a"), tmp.path().join("b"));
|
||||
fs::create_dir_all(&a).unwrap();
|
||||
fs::create_dir_all(&b).unwrap();
|
||||
fs::write(b.join("tool"), "").unwrap();
|
||||
let path_var = env::join_paths([&a, &b]).unwrap();
|
||||
assert_eq!(find_on_path("tool", Some(&path_var)), Some(b.join("tool")));
|
||||
assert_eq!(find_on_path("missing", Some(&path_var)), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn refresh_js_runtime_swaps_cached_choice() {
|
||||
let _g = env_guard();
|
||||
unsafe { env::set_var(JS_RUNTIME_ENV, "node") };
|
||||
assert_eq!(refresh_js_runtime().map(|rt| rt.kind), Some(JsRuntimeKind::Node));
|
||||
assert_eq!(resolve_js_runtime().map(|rt| rt.kind), Some(JsRuntimeKind::Node));
|
||||
|
||||
unsafe { env::set_var(JS_RUNTIME_ENV, "bun") };
|
||||
assert_eq!(resolve_js_runtime().map(|rt| rt.kind), Some(JsRuntimeKind::Node));
|
||||
assert_eq!(refresh_js_runtime().map(|rt| rt.kind), Some(JsRuntimeKind::Bun));
|
||||
|
||||
unsafe { env::remove_var(JS_RUNTIME_ENV) };
|
||||
refresh_js_runtime();
|
||||
}
|
||||
}
|
||||
|
|
@ -7,33 +7,3 @@ pub mod metadata;
|
|||
pub mod http;
|
||||
pub mod singlefile;
|
||||
pub mod font_extractor;
|
||||
pub mod text;
|
||||
pub mod js_runtime;
|
||||
pub mod deno_install;
|
||||
pub mod ytdlp_tools;
|
||||
|
||||
/// Env vars are process-global; every core test that sets resolver env vars takes this lock.
|
||||
#[cfg(test)]
|
||||
pub(crate) static RESOLVER_ENV_LOCK: std::sync::Mutex<()> = std::sync::Mutex::new(());
|
||||
|
||||
/// Writes an executable script to `path` (callers must use a fresh path each time), then
|
||||
/// waits until it can be exec'd. A child forked by a parallel test while our write fd was
|
||||
/// open keeps a copy of it until that child execs, so our own exec can fail with ETXTBSY
|
||||
/// (rust-lang/rust#114554). One exec that isn't ETXTBSY proves no writer is left, and none
|
||||
/// can appear later because our fd is already closed.
|
||||
#[cfg(all(test, unix))]
|
||||
pub(crate) fn write_script(path: &std::path::Path, body: &str) {
|
||||
use std::os::unix::fs::PermissionsExt;
|
||||
std::fs::create_dir_all(path.parent().unwrap()).unwrap();
|
||||
std::fs::write(path, body).unwrap();
|
||||
std::fs::set_permissions(path, std::fs::Permissions::from_mode(0o755)).unwrap();
|
||||
for _ in 0..200 {
|
||||
match std::process::Command::new(path).arg("--version").output() {
|
||||
Err(e) if e.kind() == std::io::ErrorKind::ExecutableFileBusy => {
|
||||
std::thread::sleep(std::time::Duration::from_millis(5));
|
||||
}
|
||||
_ => return,
|
||||
}
|
||||
}
|
||||
panic!("{} stayed busy (ETXTBSY)", path.display());
|
||||
}
|
||||
|
|
|
|||
|
|
@ -1,118 +0,0 @@
|
|||
use anyhow::{bail, Result};
|
||||
use std::path::{Path, PathBuf};
|
||||
|
||||
use crate::hash::hash_bytes;
|
||||
|
||||
/// Represents a staged text file ready to be moved into the raw store.
|
||||
#[derive(Debug)]
|
||||
pub struct StagedText {
|
||||
pub staged_path: PathBuf,
|
||||
pub hash: String,
|
||||
pub extension: String,
|
||||
pub byte_size: u64,
|
||||
}
|
||||
|
||||
/// Stages a text body (plain or Markdown) in the temp directory and computes its hash.
|
||||
///
|
||||
/// # Arguments
|
||||
/// * `body` - The raw bytes of the text content
|
||||
/// * `mime` - MIME type, must be "text/plain" or "text/markdown"
|
||||
/// * `store_path` - Root store path where temp/ subdirectory will be created
|
||||
/// * `timestamp` - Timestamp string used in the staged file name
|
||||
///
|
||||
/// # Returns
|
||||
/// * `StagedText` with the staged path, hash, extension, and byte size
|
||||
///
|
||||
/// # Errors
|
||||
/// * Rejects MIME types other than "text/plain" or "text/markdown"
|
||||
/// * IO errors during directory creation or file writing
|
||||
pub fn save(body: &[u8], mime: &str, store_path: &Path, timestamp: &str) -> Result<StagedText> {
|
||||
// Validate MIME type
|
||||
let extension = match mime {
|
||||
"text/markdown" => ".md",
|
||||
"text/plain" => ".txt",
|
||||
_ => bail!("unsupported MIME type: {mime}. Must be 'text/plain' or 'text/markdown'"),
|
||||
};
|
||||
|
||||
// Create temp directory
|
||||
let temp_dir = store_path.join("temp").join(timestamp);
|
||||
std::fs::create_dir_all(&temp_dir)?;
|
||||
|
||||
// Stage under temp/<timestamp>/<timestamp><ext>
|
||||
let staged_path = temp_dir.join(format!("{timestamp}{extension}"));
|
||||
|
||||
// Write the content
|
||||
std::fs::write(&staged_path, body)?;
|
||||
|
||||
// Compute SHA3 hash
|
||||
let hash = hash_bytes(body);
|
||||
let byte_size = body.len() as u64;
|
||||
let extension_str = extension.trim_start_matches('.').to_string();
|
||||
|
||||
Ok(StagedText {
|
||||
staged_path,
|
||||
hash,
|
||||
extension: extension_str,
|
||||
byte_size,
|
||||
})
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use tempfile::TempDir;
|
||||
|
||||
#[test]
|
||||
fn test_save_markdown() {
|
||||
let temp_dir = TempDir::new().unwrap();
|
||||
let store_path = temp_dir.path();
|
||||
let content = b"# Hello\n\nThis is markdown.";
|
||||
let mime = "text/markdown";
|
||||
|
||||
let result = save(content, mime, store_path, "2024-01-01T12-00-00.000-abc123").unwrap();
|
||||
|
||||
assert_eq!(result.extension, "md");
|
||||
assert_eq!(result.byte_size, content.len() as u64);
|
||||
assert!(result.staged_path.exists());
|
||||
assert_eq!(std::fs::read(&result.staged_path).unwrap(), content);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_save_plain_text() {
|
||||
let temp_dir = TempDir::new().unwrap();
|
||||
let store_path = temp_dir.path();
|
||||
let content = b"Plain text content";
|
||||
let mime = "text/plain";
|
||||
|
||||
let result = save(content, mime, store_path, "2024-01-01T12-00-00.000-abc123").unwrap();
|
||||
|
||||
assert_eq!(result.extension, "txt");
|
||||
assert_eq!(result.byte_size, content.len() as u64);
|
||||
assert!(result.staged_path.exists());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_save_rejects_unsupported_mime() {
|
||||
let temp_dir = TempDir::new().unwrap();
|
||||
let store_path = temp_dir.path();
|
||||
let content = b"test";
|
||||
|
||||
let result = save(content, "text/html", store_path, "2024-01-01T12-00-00.000-abc123");
|
||||
assert!(result.is_err());
|
||||
assert!(result.unwrap_err().to_string().contains("unsupported MIME type"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_save_hash_is_consistent() {
|
||||
let temp_dir = TempDir::new().unwrap();
|
||||
let store_path = temp_dir.path();
|
||||
let content = b"archivr text";
|
||||
|
||||
let result1 = save(content, "text/plain", store_path, "2024-01-01T12-00-00.000-abc123").unwrap();
|
||||
|
||||
let temp_dir2 = TempDir::new().unwrap();
|
||||
let result2 = save(content, "text/plain", temp_dir2.path(), "2024-01-01T12-00-01.000-def456").unwrap();
|
||||
|
||||
assert_eq!(result1.hash, result2.hash);
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load diff
|
|
@ -1,559 +0,0 @@
|
|||
//! yt-dlp + Deno self-update and status shared by `archivr yt-dlp` and the admin API.
|
||||
//! Sync; network via blocking reqwest.
|
||||
|
||||
use anyhow::{anyhow, bail, Context, Result};
|
||||
use serde::Serialize;
|
||||
use std::{
|
||||
env,
|
||||
path::{Path, PathBuf},
|
||||
process::Command,
|
||||
};
|
||||
|
||||
use super::deno_install;
|
||||
use super::js_runtime::{
|
||||
forced_js_runtime, path_deno, pinned_deno, probe_deno_version, probe_js_runtime_version,
|
||||
refresh_js_runtime, resolve_js_runtime_with_role, state_dir_deno, JsRuntimeRole,
|
||||
JS_RUNTIME_ENV,
|
||||
};
|
||||
use super::ytdlp::{
|
||||
forced_yt_dlp, pinned_yt_dlp, probe_version, refresh_yt_dlp, resolve_yt_dlp, state_dir,
|
||||
state_dir_yt_dlp,
|
||||
};
|
||||
|
||||
/// GitHub release metadata endpoint for the upstream yt-dlp project.
|
||||
pub const YT_DLP_LATEST_RELEASE: &str =
|
||||
"https://api.github.com/repos/yt-dlp/yt-dlp/releases/latest";
|
||||
|
||||
/// Every python zipapp starts with this shebang; used as a sanity check that we
|
||||
/// downloaded the artifact and not an HTML error page or an LFS pointer.
|
||||
const ZIPAPP_SHEBANG: &[u8] = b"#!/usr/bin/env python3";
|
||||
|
||||
/// One candidate slot of a `status` table.
|
||||
#[derive(Debug, Clone, Serialize)]
|
||||
pub struct ToolCandidate {
|
||||
/// Stable key: "force" | "env" | "state-dir" | "path".
|
||||
pub role: &'static str,
|
||||
/// Exact CLI row label.
|
||||
pub label: &'static str,
|
||||
/// `None` = empty slot (the CLI renders dashes).
|
||||
pub path: Option<String>,
|
||||
pub version: Option<String>,
|
||||
pub chosen: bool,
|
||||
/// Why the candidate can't be used: an invalid `ARCHIVR_JS_RUNTIME`, or a yt-dlp
|
||||
/// candidate that exists but whose `--version` probe fails (last stderr line).
|
||||
pub invalid: Option<String>,
|
||||
}
|
||||
|
||||
/// The candidate the resolver picked.
|
||||
#[derive(Debug, Clone, Serialize)]
|
||||
pub struct ChosenTool {
|
||||
/// `None` if the cached yt-dlp path matches no row.
|
||||
pub role: Option<&'static str>,
|
||||
/// JS only: "deno" | "node" | "bun" | "quickjs".
|
||||
pub kind: Option<&'static str>,
|
||||
pub path: Option<String>,
|
||||
pub version: Option<String>,
|
||||
}
|
||||
|
||||
/// Everything `archivr yt-dlp status` prints, as data.
|
||||
#[derive(Debug, Clone, Serialize)]
|
||||
pub struct ToolsStatus {
|
||||
/// force, env, state-dir, path-fallback (CLI order).
|
||||
pub yt_dlp: Vec<ToolCandidate>,
|
||||
pub yt_dlp_chosen: ChosenTool,
|
||||
/// force, env (ARCHIVR_DENO), state-dir, path (deno).
|
||||
pub js_runtime: Vec<ToolCandidate>,
|
||||
pub js_runtime_chosen: Option<ChosenTool>,
|
||||
pub state_dir: Option<String>,
|
||||
pub yt_dlp_target: Option<String>,
|
||||
pub yt_dlp_installed: bool,
|
||||
pub deno_target: Option<String>,
|
||||
pub deno_installed: bool,
|
||||
}
|
||||
|
||||
/// Per-component outcome of [`update_tools`]; `Ok` carries a one-line human outcome.
|
||||
pub struct UpdateReport {
|
||||
pub yt_dlp: Result<String>,
|
||||
pub deno: Result<String>,
|
||||
}
|
||||
|
||||
impl UpdateReport {
|
||||
/// Names of the failed components, in `["yt-dlp", "deno"]` order.
|
||||
pub fn failed_components(&self) -> Vec<&'static str> {
|
||||
[("yt-dlp", self.yt_dlp.is_err()), ("deno", self.deno.is_err())]
|
||||
.into_iter()
|
||||
.filter_map(|(name, failed)| failed.then_some(name))
|
||||
.collect()
|
||||
}
|
||||
}
|
||||
|
||||
/// Resolves `<state_dir>/yt-dlp/`, erroring out if there is no usable HOME.
|
||||
fn yt_dlp_state_dir() -> Result<PathBuf> {
|
||||
state_dir()
|
||||
.map(|d| d.join("yt-dlp"))
|
||||
.context("could not determine a state directory (is $HOME set?)")
|
||||
}
|
||||
|
||||
/// Asks the GitHub API for the newest yt-dlp release tag.
|
||||
fn latest_yt_dlp_version(client: &reqwest::blocking::Client) -> Result<String> {
|
||||
let body = client
|
||||
.get(YT_DLP_LATEST_RELEASE)
|
||||
.send()
|
||||
.context("failed to reach the GitHub releases API")?
|
||||
.error_for_status()
|
||||
.context("GitHub releases API returned an error")?
|
||||
.text()
|
||||
.context("failed to read the GitHub releases API response")?;
|
||||
|
||||
let json: serde_json::Value =
|
||||
serde_json::from_str(&body).context("GitHub releases API returned invalid JSON")?;
|
||||
|
||||
json.get("tag_name")
|
||||
.and_then(|t| t.as_str())
|
||||
.map(str::to_string)
|
||||
.context("GitHub releases API response had no tag_name")
|
||||
}
|
||||
|
||||
/// Installs or updates the yt-dlp zipapp in the state dir. Progress goes to `log`;
|
||||
/// `Ok` carries a one-line human outcome.
|
||||
pub fn install_yt_dlp(
|
||||
client: &reqwest::blocking::Client,
|
||||
requested_version: Option<&str>,
|
||||
log: &mut dyn FnMut(&str),
|
||||
) -> Result<String> {
|
||||
let dir = yt_dlp_state_dir()?;
|
||||
let target = dir.join("yt-dlp");
|
||||
let staging = dir.join("yt-dlp.new");
|
||||
let version_file = dir.join(".version");
|
||||
|
||||
let version = match requested_version {
|
||||
Some(v) => v.to_string(),
|
||||
None => latest_yt_dlp_version(client)?,
|
||||
};
|
||||
|
||||
// The sibling .version file is what lets us skip a ~3MB download on a
|
||||
// no-op update; the binary itself is a zipapp with no cheap version probe
|
||||
// that doesn't cost a python startup.
|
||||
let installed = std::fs::read_to_string(&version_file).ok();
|
||||
if target.is_file() && installed.as_deref().map(str::trim) == Some(version.as_str()) {
|
||||
log(&format!("yt-dlp {version} is already installed at {}", target.display()));
|
||||
return Ok(format!("yt-dlp {version} already installed at {}", target.display()));
|
||||
}
|
||||
|
||||
log(&format!("Downloading yt-dlp {version}…"));
|
||||
let url = format!("https://github.com/yt-dlp/yt-dlp/releases/download/{version}/yt-dlp");
|
||||
let bytes = client
|
||||
.get(&url)
|
||||
.send()
|
||||
.with_context(|| format!("failed to download {url}"))?
|
||||
.error_for_status()
|
||||
.with_context(|| format!("download failed — is {version} a real release tag?"))?
|
||||
.bytes()
|
||||
.context("failed to read the downloaded yt-dlp body")?;
|
||||
|
||||
if !bytes.starts_with(ZIPAPP_SHEBANG) {
|
||||
bail!(
|
||||
"downloaded artifact from {url} is not a python zipapp \
|
||||
(expected it to start with `{}`) — refusing to install it",
|
||||
String::from_utf8_lossy(ZIPAPP_SHEBANG)
|
||||
);
|
||||
}
|
||||
|
||||
std::fs::create_dir_all(&dir)
|
||||
.with_context(|| format!("failed to create {}", dir.display()))?;
|
||||
std::fs::write(&staging, &bytes)
|
||||
.with_context(|| format!("failed to write {}", staging.display()))?;
|
||||
|
||||
#[cfg(unix)]
|
||||
{
|
||||
use std::os::unix::fs::PermissionsExt;
|
||||
std::fs::set_permissions(&staging, std::fs::Permissions::from_mode(0o755))
|
||||
.with_context(|| format!("failed to chmod +x {}", staging.display()))?;
|
||||
}
|
||||
|
||||
// Atomic swap: a concurrently-running archivr sees either the whole old
|
||||
// binary or the whole new one, never a half-written file.
|
||||
std::fs::rename(&staging, &target)
|
||||
.with_context(|| format!("failed to install {}", target.display()))?;
|
||||
std::fs::write(&version_file, format!("{version}\n"))
|
||||
.with_context(|| format!("failed to record version in {}", version_file.display()))?;
|
||||
|
||||
// The zipapp is python source, not a native binary — installing it on a
|
||||
// host without python3 is legal (the server may run under a nix wrapper
|
||||
// with its own PATH) but worth flagging loudly. Kept as `warning:` (not
|
||||
// `warn:`) so the CLI's stderr is unchanged.
|
||||
let has_python = Command::new("python3")
|
||||
.arg("--version")
|
||||
.output()
|
||||
.map(|o| o.status.success())
|
||||
.unwrap_or(false);
|
||||
if !has_python {
|
||||
eprintln!(
|
||||
"warning: python3 was not found on PATH — the yt-dlp zipapp just installed \
|
||||
at {} will not run until python3 is available",
|
||||
target.display()
|
||||
);
|
||||
}
|
||||
let python_note = (!has_python)
|
||||
.then_some("; warning: python3 not found on PATH — the zipapp will not run until it is");
|
||||
|
||||
log(&format!("Installed yt-dlp {version} to {}", target.display()));
|
||||
log("archivr will now prefer it whenever it is newer than the pinned binary (ARCHIVR_YT_DLP).");
|
||||
|
||||
Ok(format!(
|
||||
"installed yt-dlp {version} to {}{}",
|
||||
target.display(),
|
||||
python_note.unwrap_or("")
|
||||
))
|
||||
}
|
||||
|
||||
/// Hint appended when the installed zipapp does not run; the zipapp is python source.
|
||||
const PYTHON_HINT: &str = "yt-dlp needs Python ≥ 3.10 on the server's PATH as `python3`";
|
||||
|
||||
/// Installs yt-dlp and Deno independently: a Deno failure never blocks the yt-dlp
|
||||
/// update (and vice versa). `Err` only if the HTTP client cannot be built.
|
||||
///
|
||||
/// With `refresh` (long-running server), the installed yt-dlp is probed with
|
||||
/// `--version` — an install that does not run is reported as a failure naming the
|
||||
/// cause — and each successful component refreshes its resolver cache, so the next
|
||||
/// yt-dlp call uses the new binary. The one-shot CLI passes `false`: it has no cache
|
||||
/// worth refreshing, and its output and probe count stay as before.
|
||||
pub fn update_tools(
|
||||
requested_yt_dlp_version: Option<&str>,
|
||||
user_agent: &str,
|
||||
refresh: bool,
|
||||
log: &mut dyn FnMut(&str),
|
||||
) -> Result<UpdateReport> {
|
||||
let client = reqwest::blocking::Client::builder()
|
||||
.user_agent(user_agent)
|
||||
.build()
|
||||
.context("failed to build an HTTP client")?;
|
||||
|
||||
let mut yt_dlp = install_yt_dlp(&client, requested_yt_dlp_version, log);
|
||||
if refresh && yt_dlp.is_ok() {
|
||||
if let Some(target) = state_dir_yt_dlp() {
|
||||
if let Err(Some(reason)) = probe_version_detail(&target) {
|
||||
yt_dlp = Err(anyhow!(unusable_install_message(&target, &reason)));
|
||||
}
|
||||
}
|
||||
refresh_yt_dlp();
|
||||
}
|
||||
let deno = deno_install::install_deno(&client, log);
|
||||
if refresh && deno.is_ok() {
|
||||
refresh_js_runtime();
|
||||
}
|
||||
Ok(UpdateReport { yt_dlp, deno })
|
||||
}
|
||||
|
||||
/// Error text for an installed yt-dlp whose `--version` probe failed.
|
||||
fn unusable_install_message(target: &Path, reason: &str) -> String {
|
||||
format!(
|
||||
"installed {} but it does not run: {reason} — {PYTHON_HINT}",
|
||||
target.display()
|
||||
)
|
||||
}
|
||||
|
||||
/// Short reason for a failed `--version` run: the last non-empty stderr line (a Python
|
||||
/// traceback ends with the actual error), else the exit status.
|
||||
fn probe_failure_reason(stderr: &[u8], status: &str) -> String {
|
||||
String::from_utf8_lossy(stderr)
|
||||
.lines()
|
||||
.map(str::trim)
|
||||
.rfind(|l| !l.is_empty())
|
||||
.map_or_else(|| format!("--version failed ({status})"), str::to_string)
|
||||
}
|
||||
|
||||
/// Runs `<binary> --version`. `Err(None)` = nothing to run (not found); `Err(Some(reason))`
|
||||
/// = the binary exists but the probe failed.
|
||||
fn probe_version_detail(binary: &Path) -> std::result::Result<String, Option<String>> {
|
||||
let out = match Command::new(binary).arg("--version").output() {
|
||||
Ok(out) => out,
|
||||
Err(e) if e.kind() == std::io::ErrorKind::NotFound => return Err(None),
|
||||
Err(e) => return Err(Some(format!("could not run: {e}"))),
|
||||
};
|
||||
if !out.status.success() {
|
||||
return Err(Some(probe_failure_reason(&out.stderr, &out.status.to_string())));
|
||||
}
|
||||
let version = String::from_utf8_lossy(&out.stdout).trim().to_string();
|
||||
if version.is_empty() {
|
||||
return Err(Some("--version printed nothing".into()));
|
||||
}
|
||||
Ok(version)
|
||||
}
|
||||
|
||||
fn display(p: &Path) -> String {
|
||||
p.display().to_string()
|
||||
}
|
||||
|
||||
/// One yt-dlp row, probing the candidate's version; a failing probe sets `invalid`.
|
||||
fn yt_row(role: &'static str, label: &'static str, path: Option<&Path>, chosen: &Path) -> ToolCandidate {
|
||||
let (version, invalid) = match path.map(probe_version_detail) {
|
||||
Some(Ok(v)) => (Some(v), None),
|
||||
Some(Err(reason)) => (None, reason),
|
||||
None => (None, None),
|
||||
};
|
||||
ToolCandidate {
|
||||
role,
|
||||
label,
|
||||
path: path.map(display),
|
||||
version,
|
||||
chosen: path == Some(chosen),
|
||||
invalid,
|
||||
}
|
||||
}
|
||||
|
||||
/// Every yt-dlp and JS runtime candidate, its version, and which one wins — the data
|
||||
/// `archivr yt-dlp status` prints. Spawns `--version` probes; call off async threads.
|
||||
pub fn tools_status() -> ToolsStatus {
|
||||
// yt-dlp: the cached choice, i.e. what this process actually runs.
|
||||
let chosen = resolve_yt_dlp();
|
||||
let state_candidate = state_dir_yt_dlp().filter(|p| p.is_file());
|
||||
let yt_dlp = vec![
|
||||
yt_row("force", "force (ARCHIVR_YT_DLP_FORCE)", forced_yt_dlp().as_deref(), &chosen),
|
||||
yt_row("env", "env (ARCHIVR_YT_DLP)", pinned_yt_dlp().as_deref(), &chosen),
|
||||
yt_row("state-dir", "state-dir", state_candidate.as_deref(), &chosen),
|
||||
yt_row("path", "path-fallback (yt-dlp)", Some(Path::new("yt-dlp")), &chosen),
|
||||
];
|
||||
let yt_dlp_chosen = match yt_dlp.iter().find(|c| c.chosen) {
|
||||
Some(c) => ChosenTool {
|
||||
role: Some(c.role),
|
||||
kind: None,
|
||||
path: Some(display(&chosen)),
|
||||
version: c.version.clone(),
|
||||
},
|
||||
None => ChosenTool {
|
||||
role: None,
|
||||
kind: None,
|
||||
path: Some(display(&chosen)),
|
||||
version: probe_version(&chosen),
|
||||
},
|
||||
};
|
||||
|
||||
// JS runtime: uncached and silent, so status never prints the resolver warnings.
|
||||
let js_chosen = resolve_js_runtime_with_role();
|
||||
let chosen_role = js_chosen.as_ref().map(|(role, _)| *role);
|
||||
let force_role = JsRuntimeRole::Forced;
|
||||
let force_row = match forced_js_runtime() {
|
||||
Ok(Some(rt)) => ToolCandidate {
|
||||
role: force_role.key(),
|
||||
label: force_role.label(),
|
||||
path: Some(rt.spec().to_string_lossy().into_owned()),
|
||||
version: probe_js_runtime_version(&rt),
|
||||
chosen: chosen_role == Some(force_role),
|
||||
invalid: None,
|
||||
},
|
||||
Ok(None) => ToolCandidate {
|
||||
role: force_role.key(),
|
||||
label: force_role.label(),
|
||||
path: None,
|
||||
version: None,
|
||||
chosen: false,
|
||||
invalid: None,
|
||||
},
|
||||
Err(reason) => ToolCandidate {
|
||||
role: force_role.key(),
|
||||
label: force_role.label(),
|
||||
path: Some(
|
||||
env::var_os(JS_RUNTIME_ENV)
|
||||
.unwrap_or_default()
|
||||
.to_string_lossy()
|
||||
.into_owned(),
|
||||
),
|
||||
version: None,
|
||||
chosen: false,
|
||||
invalid: Some(reason),
|
||||
},
|
||||
};
|
||||
let deno_row = |role: JsRuntimeRole, path: Option<PathBuf>| ToolCandidate {
|
||||
role: role.key(),
|
||||
label: role.label(),
|
||||
version: path
|
||||
.as_deref()
|
||||
.and_then(probe_deno_version)
|
||||
.map(|v| v.to_string()),
|
||||
path: path.as_deref().map(display),
|
||||
chosen: chosen_role == Some(role),
|
||||
invalid: None,
|
||||
};
|
||||
let js_runtime = vec![
|
||||
force_row,
|
||||
deno_row(JsRuntimeRole::Pinned, pinned_deno()),
|
||||
deno_row(JsRuntimeRole::StateDir, state_dir_deno().filter(|p| p.is_file())),
|
||||
deno_row(JsRuntimeRole::Path, path_deno()),
|
||||
];
|
||||
let js_runtime_chosen = js_chosen.map(|(role, rt)| ChosenTool {
|
||||
role: Some(role.key()),
|
||||
kind: Some(rt.kind.as_str()),
|
||||
path: rt.path.as_deref().map(display),
|
||||
version: js_runtime
|
||||
.iter()
|
||||
.find(|c| c.chosen)
|
||||
.and_then(|c| c.version.clone()),
|
||||
});
|
||||
|
||||
let yt_dlp_target = state_dir_yt_dlp();
|
||||
let deno_target = state_dir_deno();
|
||||
ToolsStatus {
|
||||
yt_dlp,
|
||||
yt_dlp_chosen,
|
||||
js_runtime,
|
||||
js_runtime_chosen,
|
||||
state_dir: state_dir().as_deref().map(display),
|
||||
yt_dlp_installed: yt_dlp_target.as_deref().is_some_and(Path::is_file),
|
||||
yt_dlp_target: yt_dlp_target.as_deref().map(display),
|
||||
deno_installed: deno_target.as_deref().is_some_and(Path::is_file),
|
||||
deno_target: deno_target.as_deref().map(display),
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::downloader::js_runtime::DENO_ENV;
|
||||
use crate::downloader::ytdlp::{STATE_DIR_ENV, YT_DLP_ENV, YT_DLP_FORCE_ENV};
|
||||
use anyhow::anyhow;
|
||||
|
||||
const RESOLVER_ENVS: [&str; 5] =
|
||||
[YT_DLP_FORCE_ENV, YT_DLP_ENV, STATE_DIR_ENV, JS_RUNTIME_ENV, DENO_ENV];
|
||||
|
||||
#[cfg(unix)]
|
||||
#[test]
|
||||
fn tools_status_reports_forced_and_invalid_override() {
|
||||
let _guard = crate::downloader::RESOLVER_ENV_LOCK
|
||||
.lock()
|
||||
.unwrap_or_else(|e| e.into_inner());
|
||||
for key in RESOLVER_ENVS {
|
||||
unsafe { env::remove_var(key) };
|
||||
}
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let state = tmp.path().join("state");
|
||||
let forced = tmp.path().join("forced/yt-dlp");
|
||||
crate::downloader::write_script(&forced, "#!/bin/sh\necho 2020.01.01\n");
|
||||
unsafe {
|
||||
env::set_var(STATE_DIR_ENV, &state);
|
||||
env::set_var(YT_DLP_FORCE_ENV, &forced);
|
||||
env::set_var(JS_RUNTIME_ENV, "python");
|
||||
}
|
||||
refresh_yt_dlp();
|
||||
|
||||
let status = tools_status();
|
||||
assert_eq!(status.yt_dlp[0].role, "force");
|
||||
assert!(status.yt_dlp[0].chosen);
|
||||
assert_eq!(status.yt_dlp[0].version.as_deref(), Some("2020.01.01"));
|
||||
assert_eq!(status.yt_dlp_chosen.role, Some("force"));
|
||||
let js_force = &status.js_runtime[0];
|
||||
assert!(
|
||||
js_force
|
||||
.invalid
|
||||
.as_deref()
|
||||
.is_some_and(|r| r.contains("unknown runtime python")),
|
||||
"{js_force:?}"
|
||||
);
|
||||
assert!(!js_force.chosen);
|
||||
assert!(!status.yt_dlp_installed);
|
||||
assert!(status
|
||||
.yt_dlp_target
|
||||
.as_deref()
|
||||
.is_some_and(|t| t.ends_with("yt-dlp/yt-dlp")));
|
||||
let json = serde_json::to_value(&status).unwrap();
|
||||
for key in ["yt_dlp", "js_runtime", "state_dir", "deno_target"] {
|
||||
assert!(json.get(key).is_some(), "missing {key}");
|
||||
}
|
||||
|
||||
for key in RESOLVER_ENVS {
|
||||
unsafe { env::remove_var(key) };
|
||||
}
|
||||
refresh_yt_dlp();
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn probe_failure_reason_prefers_last_stderr_line() {
|
||||
assert_eq!(
|
||||
probe_failure_reason(
|
||||
b"Traceback (most recent call last):\n File \"yt_dlp/__main__.py\", line 13\n\
|
||||
ImportError: You are using an unsupported version of Python. Only Python \
|
||||
versions 3.10 and above are supported by yt-dlp\n\n",
|
||||
"exit status: 1"
|
||||
),
|
||||
"ImportError: You are using an unsupported version of Python. Only Python \
|
||||
versions 3.10 and above are supported by yt-dlp"
|
||||
);
|
||||
assert_eq!(
|
||||
probe_failure_reason(b"\n boom: too old \n", "exit status: 1"),
|
||||
"boom: too old"
|
||||
);
|
||||
assert_eq!(
|
||||
probe_failure_reason(b" \n", "exit status: 2"),
|
||||
"--version failed (exit status: 2)"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn unusable_install_message_names_cause_and_hint() {
|
||||
let msg = unusable_install_message(Path::new("/s/yt-dlp/yt-dlp"), "Only Python 3.10+");
|
||||
assert!(msg.contains("/s/yt-dlp/yt-dlp"), "{msg}");
|
||||
assert!(msg.contains("Only Python 3.10+"), "{msg}");
|
||||
assert!(msg.contains("Python ≥ 3.10"), "{msg}");
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
#[test]
|
||||
fn probe_version_detail_classifies_outcomes() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let ok = tmp.path().join("ok");
|
||||
crate::downloader::write_script(&ok, "#!/bin/sh\necho 2024.01.01\n");
|
||||
assert_eq!(probe_version_detail(&ok), Ok("2024.01.01".into()));
|
||||
let bad = tmp.path().join("bad");
|
||||
crate::downloader::write_script(
|
||||
&bad,
|
||||
"#!/bin/sh\necho 'Only Python versions 3.10 and above are supported' >&2\nexit 1\n",
|
||||
);
|
||||
assert_eq!(
|
||||
probe_version_detail(&bad),
|
||||
Err(Some("Only Python versions 3.10 and above are supported".into()))
|
||||
);
|
||||
assert_eq!(probe_version_detail(&tmp.path().join("missing")), Err(None));
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
#[test]
|
||||
fn tools_status_reports_unusable_candidate_reason() {
|
||||
let _guard = crate::downloader::RESOLVER_ENV_LOCK
|
||||
.lock()
|
||||
.unwrap_or_else(|e| e.into_inner());
|
||||
for key in RESOLVER_ENVS {
|
||||
unsafe { env::remove_var(key) };
|
||||
}
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let pinned = tmp.path().join("pinned/yt-dlp");
|
||||
crate::downloader::write_script(&pinned, "#!/bin/sh\necho 'boom: too old' >&2\nexit 1\n");
|
||||
unsafe {
|
||||
env::set_var(STATE_DIR_ENV, tmp.path().join("state"));
|
||||
env::set_var(YT_DLP_ENV, &pinned);
|
||||
}
|
||||
refresh_yt_dlp();
|
||||
|
||||
let status = tools_status();
|
||||
let env_row = &status.yt_dlp[1];
|
||||
assert_eq!(env_row.role, "env");
|
||||
assert_eq!(env_row.version, None);
|
||||
assert_eq!(env_row.invalid.as_deref(), Some("boom: too old"));
|
||||
|
||||
for key in RESOLVER_ENVS {
|
||||
unsafe { env::remove_var(key) };
|
||||
}
|
||||
refresh_yt_dlp();
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn failed_components_lists_only_failures() {
|
||||
let report = |y: bool, d: bool| UpdateReport {
|
||||
yt_dlp: if y { Ok("ok".into()) } else { Err(anyhow!("boom")) },
|
||||
deno: if d { Ok("ok".into()) } else { Err(anyhow!("boom")) },
|
||||
};
|
||||
assert!(report(true, true).failed_components().is_empty());
|
||||
assert_eq!(report(false, true).failed_components(), ["yt-dlp"]);
|
||||
assert_eq!(report(true, false).failed_components(), ["deno"]);
|
||||
assert_eq!(report(false, false).failed_components(), ["yt-dlp", "deno"]);
|
||||
}
|
||||
}
|
||||
|
|
@ -1,109 +0,0 @@
|
|||
//! Env-var resolution helpers shared by the summary providers and the local
|
||||
//! transcription engines. External tools are configured by `ARCHIVR_*` env
|
||||
//! vars only, never TOML.
|
||||
|
||||
use anyhow::{Result, bail};
|
||||
use std::{
|
||||
env,
|
||||
path::{Path, PathBuf},
|
||||
};
|
||||
|
||||
/// Reads a required env var, failing with the *exact variable name* so the
|
||||
/// server can hand a caller an actionable 400 rather than "not configured".
|
||||
pub(crate) fn required_env(name: &str) -> Result<String> {
|
||||
match env::var(name) {
|
||||
Ok(v) if !v.trim().is_empty() => Ok(v),
|
||||
_ => bail!("missing required environment variable: {name}"),
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn env_or(name: &str, default: &str) -> String {
|
||||
env::var(name)
|
||||
.ok()
|
||||
.filter(|v| !v.trim().is_empty())
|
||||
.unwrap_or_else(|| default.to_string())
|
||||
}
|
||||
|
||||
pub(crate) fn optional_env(name: &str) -> Option<String> {
|
||||
env::var(name).ok().filter(|v| !v.trim().is_empty())
|
||||
}
|
||||
|
||||
pub(crate) fn env_timeout(name: &str, default: u64) -> u64 {
|
||||
env::var(name)
|
||||
.ok()
|
||||
.and_then(|v| v.trim().parse::<u64>().ok())
|
||||
.filter(|v| *v > 0)
|
||||
.unwrap_or(default)
|
||||
}
|
||||
|
||||
/// Resolve a CLI executable path.
|
||||
///
|
||||
/// Priority: `env_name` override → first `well_known_absolute` path that
|
||||
/// exists → `HOME/.local/bin/<bare>` if it exists → bare name (relies on the
|
||||
/// server's PATH). The macOS defaults matter for `codex`, which the ChatGPT
|
||||
/// desktop app installs at `/Applications/ChatGPT.app/Contents/Resources/codex`
|
||||
/// and does not add to PATH.
|
||||
pub(crate) fn resolve_cli(env_name: &str, well_known_absolute: &[&str], bare: &str) -> PathBuf {
|
||||
if let Some(explicit) = optional_env(env_name) {
|
||||
return PathBuf::from(explicit);
|
||||
}
|
||||
for candidate in well_known_absolute {
|
||||
let p = Path::new(candidate);
|
||||
if p.is_file() {
|
||||
return p.to_path_buf();
|
||||
}
|
||||
}
|
||||
if let Some(home) = env::var_os("HOME") {
|
||||
let mut p = PathBuf::from(home);
|
||||
p.push(".local/bin");
|
||||
p.push(bare);
|
||||
if p.is_file() {
|
||||
return p;
|
||||
}
|
||||
}
|
||||
PathBuf::from(bare)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
static ENV_LOCK: std::sync::Mutex<()> = std::sync::Mutex::new(());
|
||||
const VAR: &str = "ARCHIVR_TEST_RESOLVE_CLI";
|
||||
|
||||
#[test]
|
||||
fn resolve_cli_prefers_env_then_absolute_then_bare() {
|
||||
let _guard = ENV_LOCK.lock().unwrap_or_else(|e| e.into_inner());
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let absolute = dir.path().join("tool");
|
||||
std::fs::write(&absolute, b"").unwrap();
|
||||
let absolute_str = absolute.to_str().unwrap();
|
||||
let bare = "archivr-test-resolve-cli-surely-not-installed";
|
||||
|
||||
unsafe { env::set_var(VAR, "/explicit/tool") };
|
||||
assert_eq!(
|
||||
resolve_cli(VAR, &[absolute_str], bare),
|
||||
PathBuf::from("/explicit/tool")
|
||||
);
|
||||
|
||||
unsafe { env::remove_var(VAR) };
|
||||
assert_eq!(resolve_cli(VAR, &["/nonexistent/x", absolute_str], bare), absolute);
|
||||
assert_eq!(
|
||||
resolve_cli(VAR, &["/nonexistent/x"], bare),
|
||||
PathBuf::from(bare)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn env_timeout_ignores_zero_and_garbage() {
|
||||
let _guard = ENV_LOCK.lock().unwrap_or_else(|e| e.into_inner());
|
||||
const T: &str = "ARCHIVR_TEST_ENV_TIMEOUT";
|
||||
unsafe { env::set_var(T, "0") };
|
||||
assert_eq!(env_timeout(T, 7), 7);
|
||||
unsafe { env::set_var(T, "abc") };
|
||||
assert_eq!(env_timeout(T, 7), 7);
|
||||
unsafe { env::set_var(T, " 12 ") };
|
||||
assert_eq!(env_timeout(T, 7), 12);
|
||||
unsafe { env::remove_var(T) };
|
||||
}
|
||||
}
|
||||
|
|
@ -4,9 +4,3 @@ pub mod database;
|
|||
pub mod downloader;
|
||||
pub mod hash;
|
||||
pub mod twitter;
|
||||
pub mod summarizer;
|
||||
pub mod subtitles;
|
||||
pub mod thread_title;
|
||||
pub mod transcriber;
|
||||
pub(crate) mod env_config;
|
||||
pub(crate) mod process;
|
||||
|
|
|
|||
|
|
@ -1,381 +0,0 @@
|
|||
//! Subprocess runner with a wall-clock timeout.
|
||||
//!
|
||||
//! `archivr-core` deliberately has no async runtime and the tree carries no
|
||||
//! `wait_timeout` dependency, so the timeout is enforced by structure: stdout
|
||||
//! and stderr are drained on their own threads (a chatty child must never
|
||||
//! block on a full pipe buffer), stdin is written on a third thread (a large
|
||||
//! prompt can exceed the pipe buffer), and the calling thread polls
|
||||
//! `try_wait` until the child exits or the deadline passes, then kills it.
|
||||
|
||||
use anyhow::{Context, Result, anyhow};
|
||||
use std::{
|
||||
ffi::OsString,
|
||||
io::{Read, Write},
|
||||
path::Path,
|
||||
process::{Child, Command, Stdio},
|
||||
sync::mpsc,
|
||||
thread,
|
||||
time::{Duration, Instant},
|
||||
};
|
||||
|
||||
/// Bytes of stderr kept for diagnostics.
|
||||
const STDERR_TAIL_BYTES: usize = 4096;
|
||||
/// Characters of the stderr tail quoted in a non-zero-exit error.
|
||||
const EXIT_ERROR_STDERR_CHARS: usize = 400;
|
||||
const POLL_INTERVAL: Duration = Duration::from_millis(50);
|
||||
/// How long to wait for the pipe readers after the direct child exits before
|
||||
/// assuming a grandchild holds the pipes and killing the process group.
|
||||
pub(crate) const READER_GRACE: Duration = Duration::from_secs(2);
|
||||
|
||||
/// Puts the child in its own process group (unix) so a timeout can kill the
|
||||
/// whole tree, including grandchildren that inherited the output pipes.
|
||||
pub(crate) fn isolate_process_group(cmd: &mut Command) {
|
||||
#[cfg(unix)]
|
||||
{
|
||||
use std::os::unix::process::CommandExt;
|
||||
cmd.process_group(0);
|
||||
}
|
||||
#[cfg(not(unix))]
|
||||
let _ = cmd;
|
||||
}
|
||||
|
||||
/// SIGKILLs the process group led by `pid` (spawned via
|
||||
/// [`isolate_process_group`]). Best effort; a missing group (`ESRCH`) is not
|
||||
/// an error. Calls `kill(2)` directly: slim runtime images ship no `kill`
|
||||
/// binary.
|
||||
pub(crate) fn kill_process_group(pid: u32) {
|
||||
#[cfg(unix)]
|
||||
{
|
||||
// kill(0, ..) hits our own group and kill(-1, ..) every process we may
|
||||
// signal; a pid that doesn't fit pid_t can't be a real child either.
|
||||
let Ok(pgid) = libc::pid_t::try_from(pid) else {
|
||||
return;
|
||||
};
|
||||
if pgid <= 1 {
|
||||
return;
|
||||
}
|
||||
// SAFETY: kill(2) takes plain integers and touches no memory of ours;
|
||||
// a negative pid targets the process group `pgid`.
|
||||
let _ = unsafe { libc::kill(-pgid, libc::SIGKILL) };
|
||||
}
|
||||
#[cfg(not(unix))]
|
||||
let _ = pid;
|
||||
}
|
||||
|
||||
/// Kills the child's whole process group and reaps the direct child.
|
||||
pub(crate) fn kill_tree(child: &mut Child) {
|
||||
kill_process_group(child.id());
|
||||
let _ = child.kill();
|
||||
let _ = child.wait();
|
||||
}
|
||||
|
||||
/// Receives a reader result after the direct child exited: waits up to
|
||||
/// `min(grace, budget)`, then kills the process group (a grandchild holding
|
||||
/// the pipe) and waits one more grace period. `None` if still not done.
|
||||
pub(crate) fn recv_after_exit<T>(rx: &mpsc::Receiver<T>, pid: u32, budget: Duration) -> Option<T> {
|
||||
if let Ok(v) = rx.recv_timeout(READER_GRACE.min(budget)) {
|
||||
return Some(v);
|
||||
}
|
||||
kill_process_group(pid);
|
||||
rx.recv_timeout(READER_GRACE).ok()
|
||||
}
|
||||
|
||||
#[derive(Debug)]
|
||||
pub(crate) struct ProcessOutput {
|
||||
pub stdout: String,
|
||||
/// Last 4 KiB of stderr, lossy UTF-8.
|
||||
#[allow(dead_code)]
|
||||
pub stderr_tail: String,
|
||||
}
|
||||
|
||||
/// Sentinel at the root of a timeout error, so callers can recognise a timeout
|
||||
/// without string matching (see [`is_process_timeout`]).
|
||||
#[derive(Debug)]
|
||||
pub(crate) struct ProcessTimedOut {
|
||||
pub secs: u64,
|
||||
}
|
||||
|
||||
impl std::fmt::Display for ProcessTimedOut {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
write!(f, "timed out after {}s", self.secs)
|
||||
}
|
||||
}
|
||||
|
||||
impl std::error::Error for ProcessTimedOut {}
|
||||
|
||||
/// True when `error` came from a [`run_with_timeout`] deadline, however many
|
||||
/// context layers have been added on top since.
|
||||
pub(crate) fn is_process_timeout(error: &anyhow::Error) -> bool {
|
||||
error
|
||||
.chain()
|
||||
.find_map(|c| c.downcast_ref::<ProcessTimedOut>())
|
||||
.or_else(|| error.downcast_ref::<ProcessTimedOut>())
|
||||
.is_some()
|
||||
}
|
||||
|
||||
/// Spawns `executable args…`, optionally writes `stdin`, drains stdout and
|
||||
/// stderr on their own threads, and kills the child if it is still running at
|
||||
/// `timeout`.
|
||||
///
|
||||
/// - Non-zero exit → `Err("{exe} exited with {status}: {last 400 chars of stderr}")`.
|
||||
/// - Timeout → an error whose root is [`ProcessTimedOut`] with the message
|
||||
/// `"{exe} timed out after {secs}s"`.
|
||||
pub(crate) fn run_with_timeout(
|
||||
executable: &Path,
|
||||
args: &[OsString],
|
||||
stdin: Option<&str>,
|
||||
timeout: Duration,
|
||||
) -> Result<ProcessOutput> {
|
||||
let exe = executable.display().to_string();
|
||||
let started = Instant::now();
|
||||
let timeout_error = || {
|
||||
let secs = timeout.as_secs().max(1);
|
||||
anyhow::Error::new(ProcessTimedOut { secs }).context(format!("{exe} timed out after {secs}s"))
|
||||
};
|
||||
|
||||
let mut command = Command::new(executable);
|
||||
command
|
||||
.args(args)
|
||||
.stdin(if stdin.is_some() {
|
||||
Stdio::piped()
|
||||
} else {
|
||||
Stdio::null()
|
||||
})
|
||||
.stdout(Stdio::piped())
|
||||
.stderr(Stdio::piped());
|
||||
isolate_process_group(&mut command);
|
||||
let mut child = command
|
||||
.spawn()
|
||||
.with_context(|| format!("failed to spawn {exe}"))?;
|
||||
let pid = child.id();
|
||||
|
||||
if let Some(input) = stdin {
|
||||
let mut pipe = child
|
||||
.stdin
|
||||
.take()
|
||||
.ok_or_else(|| anyhow!("failed to open stdin for {exe}"))?;
|
||||
let owned = input.to_string();
|
||||
thread::spawn(move || {
|
||||
let _ = pipe.write_all(owned.as_bytes());
|
||||
// Dropping the pipe closes it, which tells the child input is complete.
|
||||
});
|
||||
}
|
||||
|
||||
let mut stdout = child
|
||||
.stdout
|
||||
.take()
|
||||
.ok_or_else(|| anyhow!("failed to open stdout for {exe}"))?;
|
||||
let (out_tx, out_rx) = mpsc::channel();
|
||||
thread::spawn(move || {
|
||||
let mut buf = String::new();
|
||||
let res = stdout.read_to_string(&mut buf).map(|_| buf);
|
||||
let _ = out_tx.send(res);
|
||||
});
|
||||
|
||||
let mut stderr = child
|
||||
.stderr
|
||||
.take()
|
||||
.ok_or_else(|| anyhow!("failed to open stderr for {exe}"))?;
|
||||
let (err_tx, err_rx) = mpsc::channel();
|
||||
thread::spawn(move || {
|
||||
let mut tail: Vec<u8> = Vec::new();
|
||||
let mut chunk = [0u8; 8192];
|
||||
loop {
|
||||
match stderr.read(&mut chunk) {
|
||||
Ok(0) | Err(_) => break,
|
||||
Ok(n) => {
|
||||
tail.extend_from_slice(&chunk[..n]);
|
||||
if tail.len() > STDERR_TAIL_BYTES {
|
||||
let excess = tail.len() - STDERR_TAIL_BYTES;
|
||||
tail.drain(..excess);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
let _ = err_tx.send(String::from_utf8_lossy(&tail).into_owned());
|
||||
});
|
||||
|
||||
let status = loop {
|
||||
match child.try_wait() {
|
||||
Ok(Some(status)) => break status,
|
||||
Ok(None) => {}
|
||||
Err(e) => {
|
||||
kill_tree(&mut child);
|
||||
return Err(anyhow::Error::new(e).context(format!("failed to wait for {exe}")));
|
||||
}
|
||||
}
|
||||
if started.elapsed() >= timeout {
|
||||
// Kill the whole group so grandchildren release the pipes too.
|
||||
kill_tree(&mut child);
|
||||
return Err(timeout_error());
|
||||
}
|
||||
thread::sleep(POLL_INTERVAL);
|
||||
};
|
||||
|
||||
// The child has exited, but a grandchild that inherited the pipes can keep
|
||||
// them open; give the readers a short grace, then kill the group.
|
||||
let remaining = || timeout.saturating_sub(started.elapsed());
|
||||
let collected = match recv_after_exit(&out_rx, pid, remaining()) {
|
||||
Some(res) => res.with_context(|| format!("failed to read stdout of {exe}"))?,
|
||||
None => return Err(timeout_error()),
|
||||
};
|
||||
let Some(stderr_tail) = recv_after_exit(&err_rx, pid, remaining()) else {
|
||||
return Err(timeout_error());
|
||||
};
|
||||
|
||||
if !status.success() {
|
||||
anyhow::bail!(
|
||||
"{exe} exited with {status}: {}",
|
||||
last_chars(stderr_tail.trim(), EXIT_ERROR_STDERR_CHARS)
|
||||
);
|
||||
}
|
||||
Ok(ProcessOutput {
|
||||
stdout: collected,
|
||||
stderr_tail,
|
||||
})
|
||||
}
|
||||
|
||||
fn last_chars(s: &str, max: usize) -> String {
|
||||
let count = s.chars().count();
|
||||
if count <= max {
|
||||
return s.to_string();
|
||||
}
|
||||
let tail: String = s.chars().skip(count - max).collect();
|
||||
format!("…{tail}")
|
||||
}
|
||||
|
||||
/// Shared by process-group tests here and in `downloader::ytdlp`.
|
||||
#[cfg(all(test, unix))]
|
||||
pub(crate) mod test_support {
|
||||
use std::{path::Path, process::Command, time::{Duration, Instant}};
|
||||
|
||||
/// Shell snippet: start a background `sleep 30` and record its pid in `pid_file`.
|
||||
pub(crate) fn spawn_grandchild_snippet(pid_file: &Path) -> String {
|
||||
format!("sleep 30 & echo $! > '{}'; ", pid_file.display())
|
||||
}
|
||||
|
||||
/// Reads the pid written by [`spawn_grandchild_snippet`] and asserts the
|
||||
/// process disappears within a few seconds (allowing init to reap it).
|
||||
pub(crate) fn assert_grandchild_gone(pid_file: &Path) {
|
||||
let pid = std::fs::read_to_string(pid_file).unwrap().trim().to_string();
|
||||
assert!(!pid.is_empty(), "grandchild pid not recorded");
|
||||
let deadline = Instant::now() + Duration::from_secs(5);
|
||||
loop {
|
||||
let alive = Command::new("kill")
|
||||
.args(["-0", &pid])
|
||||
.stderr(std::process::Stdio::null())
|
||||
.status()
|
||||
.map(|s| s.success())
|
||||
.unwrap_or(false);
|
||||
if !alive {
|
||||
return;
|
||||
}
|
||||
if Instant::now() >= deadline {
|
||||
let _ = Command::new("kill").args(["-KILL", &pid]).status();
|
||||
panic!("grandchild {pid} survived the timeout kill");
|
||||
}
|
||||
std::thread::sleep(Duration::from_millis(50));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn os(args: &[&str]) -> Vec<OsString> {
|
||||
args.iter().map(OsString::from).collect()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn run_with_timeout_drains_large_stderr_without_deadlock() {
|
||||
let out = run_with_timeout(
|
||||
Path::new("sh"),
|
||||
&os(&["-c", "head -c 1000000 /dev/zero | tr '\\0' x >&2; echo ok"]),
|
||||
None,
|
||||
Duration::from_secs(30),
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(out.stdout, "ok\n");
|
||||
assert_eq!(out.stderr_tail.len(), STDERR_TAIL_BYTES);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn run_with_timeout_kills_overrunning_child_and_marks_timeout() {
|
||||
let started = Instant::now();
|
||||
let err = run_with_timeout(
|
||||
Path::new("sleep"),
|
||||
&os(&["30"]),
|
||||
None,
|
||||
Duration::from_secs(1),
|
||||
)
|
||||
.unwrap_err();
|
||||
assert!(is_process_timeout(&err), "{err:#}");
|
||||
assert!(format!("{err:#}").contains("timed out after 1s"), "{err:#}");
|
||||
assert!(started.elapsed() < Duration::from_secs(5));
|
||||
// Still recognisable under further context layers.
|
||||
let wrapped = err.context("outer").context("outermost");
|
||||
assert!(is_process_timeout(&wrapped));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn run_with_timeout_reports_nonzero_exit_with_stderr_tail() {
|
||||
let err = run_with_timeout(
|
||||
Path::new("sh"),
|
||||
&os(&["-c", "echo first-line >&2; echo boom-at-the-end >&2; exit 3"]),
|
||||
None,
|
||||
Duration::from_secs(30),
|
||||
)
|
||||
.unwrap_err();
|
||||
let msg = format!("{err:#}");
|
||||
assert!(msg.contains("exited with"), "{msg}");
|
||||
assert!(msg.contains("boom-at-the-end"), "{msg}");
|
||||
assert!(!is_process_timeout(&err));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn run_with_timeout_round_trips_stdin() {
|
||||
let out = run_with_timeout(
|
||||
Path::new("cat"),
|
||||
&[],
|
||||
Some("prompt text"),
|
||||
Duration::from_secs(30),
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(out.stdout, "prompt text");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn last_chars_keeps_the_tail() {
|
||||
assert_eq!(last_chars("abc", 5), "abc");
|
||||
assert_eq!(last_chars("abcdef", 3), "…def");
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
#[test]
|
||||
fn run_with_timeout_kills_grandchildren_on_timeout() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let pid_file = dir.path().join("grandchild.pid");
|
||||
let script = format!("{}sleep 30", test_support::spawn_grandchild_snippet(&pid_file));
|
||||
let started = Instant::now();
|
||||
let err = run_with_timeout(Path::new("sh"), &os(&["-c", &script]), None, Duration::from_secs(1))
|
||||
.unwrap_err();
|
||||
assert!(is_process_timeout(&err), "{err:#}");
|
||||
assert!(started.elapsed() < Duration::from_secs(5));
|
||||
test_support::assert_grandchild_gone(&pid_file);
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
#[test]
|
||||
fn run_with_timeout_does_not_wait_out_budget_for_pipe_holding_grandchild() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let pid_file = dir.path().join("grandchild.pid");
|
||||
let script = format!("{}echo done", test_support::spawn_grandchild_snippet(&pid_file));
|
||||
let started = Instant::now();
|
||||
let out = run_with_timeout(Path::new("sh"), &os(&["-c", &script]), None, Duration::from_secs(60))
|
||||
.unwrap();
|
||||
assert_eq!(out.stdout, "done\n");
|
||||
assert!(started.elapsed() < Duration::from_secs(10), "{:?}", started.elapsed());
|
||||
test_support::assert_grandchild_gone(&pid_file);
|
||||
}
|
||||
}
|
||||
|
|
@ -1,844 +0,0 @@
|
|||
//! Subtitle artifacts: archiving staged yt-dlp subtitle files, registering
|
||||
//! them as `subtitle` artifacts, fetching them on demand for existing entries,
|
||||
//! ranking tracks, and reducing VTT/SRT to a plain transcript for summaries.
|
||||
|
||||
use anyhow::{anyhow, Context, Result};
|
||||
use regex::Regex;
|
||||
use rusqlite::{Connection, Transaction, TransactionBehavior};
|
||||
use std::{
|
||||
fs,
|
||||
path::{Path, PathBuf},
|
||||
sync::LazyLock,
|
||||
};
|
||||
use uuid::Uuid;
|
||||
|
||||
use crate::archive::ArchivePaths;
|
||||
use crate::capture;
|
||||
use crate::database::{self, BlobRecord, NewArtifact};
|
||||
use crate::downloader::store;
|
||||
use crate::downloader::ytdlp::{self, language_base, StagedSubtitle, SubtitleKind};
|
||||
|
||||
pub const SUBTITLE_ARTIFACT_ROLE: &str = "subtitle";
|
||||
pub const SUBTITLE_ORIGIN_CAPTURE: &str = "capture";
|
||||
pub const SUBTITLE_ORIGIN_SUMMARY_FETCH: &str = "summary_fetch";
|
||||
/// Origin of a track produced by local transcription (`kind: "transcribed"`).
|
||||
pub const SUBTITLE_ORIGIN_TRANSCRIPTION: &str = "transcription";
|
||||
|
||||
/// Result of [`fetch_subtitles_for_entry`].
|
||||
#[derive(Debug, Clone, Default, PartialEq, Eq)]
|
||||
pub struct SubtitleFetchOutcome {
|
||||
/// Artifact rows inserted by this call.
|
||||
pub added: usize,
|
||||
/// The video's original language, from a successful metadata probe or
|
||||
/// else from existing subtitle artifacts' metadata.
|
||||
pub original_language: Option<String>,
|
||||
}
|
||||
|
||||
/// Subtitle file formats archivr keeps.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum SubtitleFormat {
|
||||
Vtt,
|
||||
Srt,
|
||||
}
|
||||
|
||||
impl SubtitleFormat {
|
||||
/// Detects the format from a file extension (with or without the dot),
|
||||
/// falling back to the MIME type. Case-insensitive.
|
||||
pub fn detect(extension: &str, mime: &str) -> Option<Self> {
|
||||
let ext = extension.trim_start_matches('.').to_ascii_lowercase();
|
||||
match ext.as_str() {
|
||||
"vtt" => return Some(SubtitleFormat::Vtt),
|
||||
"srt" => return Some(SubtitleFormat::Srt),
|
||||
_ => {}
|
||||
}
|
||||
let mime = mime.split(';').next().unwrap_or("").trim().to_ascii_lowercase();
|
||||
match mime.as_str() {
|
||||
"text/vtt" => Some(SubtitleFormat::Vtt),
|
||||
"application/x-subrip" => Some(SubtitleFormat::Srt),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
pub fn mime(self) -> &'static str {
|
||||
match self {
|
||||
SubtitleFormat::Vtt => "text/vtt",
|
||||
SubtitleFormat::Srt => "application/x-subrip",
|
||||
}
|
||||
}
|
||||
|
||||
pub fn extension(self) -> &'static str {
|
||||
match self {
|
||||
SubtitleFormat::Vtt => "vtt",
|
||||
SubtitleFormat::Srt => "srt",
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// A subtitle file already moved into the content-addressed `raw/` store.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct ArchivedSubtitle {
|
||||
/// Store-relative path, e.g. `raw/a/b/<hash>.vtt`.
|
||||
pub raw_relpath: PathBuf,
|
||||
pub language: String,
|
||||
pub kind: SubtitleKind,
|
||||
pub format: SubtitleFormat,
|
||||
pub original_language: Option<String>,
|
||||
}
|
||||
|
||||
/// Track description parsed from a `subtitle` artifact's `metadata_json`.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct SubtitleTrackMeta {
|
||||
pub language: String,
|
||||
pub kind: SubtitleKind,
|
||||
pub original_language: Option<String>,
|
||||
}
|
||||
|
||||
/// Moves staged subtitle files into `raw/`. Files that fail to archive or have
|
||||
/// an unsupported format are logged and skipped — subtitles never fail a capture.
|
||||
pub fn archive_staged_subtitles(
|
||||
store_path: &Path,
|
||||
staged: Vec<StagedSubtitle>,
|
||||
) -> Vec<ArchivedSubtitle> {
|
||||
let mut archived = Vec::with_capacity(staged.len());
|
||||
for sub in staged {
|
||||
let Some(format) = SubtitleFormat::detect(&sub.format, "") else {
|
||||
eprintln!(
|
||||
"warn: archive subtitle {}: unsupported format {}",
|
||||
sub.path.display(),
|
||||
sub.format
|
||||
);
|
||||
continue;
|
||||
};
|
||||
match store::archive_staged_file(&sub.path, store_path) {
|
||||
Ok(raw_relpath) => archived.push(ArchivedSubtitle {
|
||||
raw_relpath,
|
||||
language: sub.language,
|
||||
kind: sub.kind,
|
||||
format,
|
||||
original_language: sub.original_language,
|
||||
}),
|
||||
Err(e) => eprintln!("warn: archive subtitle {}: {e:#}", sub.path.display()),
|
||||
}
|
||||
}
|
||||
archived
|
||||
}
|
||||
|
||||
/// Registers archived subtitles as `subtitle` artifacts of `entry_id`.
|
||||
///
|
||||
/// Runs in one `BEGIN IMMEDIATE` transaction so concurrent registrations of
|
||||
/// the same content serialize; an `(entry, subtitle, blob)` that already
|
||||
/// exists is skipped. Returns the number of artifact rows inserted.
|
||||
pub fn register_subtitle_artifacts(
|
||||
conn: &Connection,
|
||||
store_path: &Path,
|
||||
entry_id: i64,
|
||||
subtitles: &[ArchivedSubtitle],
|
||||
origin: &str,
|
||||
) -> Result<usize> {
|
||||
let rows: Vec<(&ArchivedSubtitle, serde_json::Value)> = subtitles
|
||||
.iter()
|
||||
.map(|sub| {
|
||||
let metadata = serde_json::json!({
|
||||
"language": sub.language,
|
||||
"kind": sub.kind.as_str(),
|
||||
"format": sub.format.extension(),
|
||||
"original_language": sub.original_language,
|
||||
"origin": origin,
|
||||
});
|
||||
(sub, metadata)
|
||||
})
|
||||
.collect();
|
||||
insert_subtitle_rows(conn, store_path, entry_id, &rows)
|
||||
}
|
||||
|
||||
/// Registers a locally transcribed track (origin `transcription`), recording
|
||||
/// the engine kind and model. A model given as a filesystem path is stored as
|
||||
/// its file name only, so no host path is persisted. Same transaction and
|
||||
/// dedup rules as [`register_subtitle_artifacts`].
|
||||
pub fn register_transcript_artifact(
|
||||
conn: &Connection,
|
||||
store_path: &Path,
|
||||
entry_id: i64,
|
||||
sub: &ArchivedSubtitle,
|
||||
engine: &str,
|
||||
model: &str,
|
||||
) -> Result<usize> {
|
||||
let metadata = serde_json::json!({
|
||||
"language": sub.language,
|
||||
"kind": sub.kind.as_str(),
|
||||
"format": sub.format.extension(),
|
||||
"original_language": sub.original_language,
|
||||
"origin": SUBTITLE_ORIGIN_TRANSCRIPTION,
|
||||
"engine": engine,
|
||||
"model": sanitize_model_name(model),
|
||||
});
|
||||
insert_subtitle_rows(conn, store_path, entry_id, &[(sub, metadata)])
|
||||
}
|
||||
|
||||
/// Reduces a model that is a filesystem path (contains `\`, is absolute, or
|
||||
/// exists) to its file name. Hugging Face ids such as
|
||||
/// `nvidia/parakeet-tdt-0.6b-v3` are kept as-is.
|
||||
pub(crate) fn sanitize_model_name(model: &str) -> String {
|
||||
let path = Path::new(model);
|
||||
if model.contains('\\') || path.is_absolute() || path.exists() {
|
||||
let name = model.rsplit(['/', '\\']).next().unwrap_or(model);
|
||||
if !name.is_empty() {
|
||||
return name.to_string();
|
||||
}
|
||||
}
|
||||
model.to_string()
|
||||
}
|
||||
|
||||
/// Inserts one `subtitle` artifact per row with the given metadata, in one
|
||||
/// `BEGIN IMMEDIATE` transaction; rows whose blob is already a subtitle of
|
||||
/// the entry, or whose file can't be stat'ed, are skipped. Refreshes the
|
||||
/// entry's cached bytes and returns the number of rows inserted.
|
||||
fn insert_subtitle_rows(
|
||||
conn: &Connection,
|
||||
store_path: &Path,
|
||||
entry_id: i64,
|
||||
rows: &[(&ArchivedSubtitle, serde_json::Value)],
|
||||
) -> Result<usize> {
|
||||
if rows.is_empty() {
|
||||
return Ok(0);
|
||||
}
|
||||
|
||||
// No transaction is open on `conn` at any call site, so new_unchecked is safe.
|
||||
let tx = Transaction::new_unchecked(conn, TransactionBehavior::Immediate)?;
|
||||
let mut inserted = 0;
|
||||
for (sub, metadata) in rows {
|
||||
let relpath = sub.raw_relpath.to_string_lossy().replace('\\', "/");
|
||||
let sha256 = sub
|
||||
.raw_relpath
|
||||
.file_stem()
|
||||
.and_then(|s| s.to_str())
|
||||
.with_context(|| format!("subtitle path has no hash stem: {relpath}"))?
|
||||
.to_string();
|
||||
let byte_size = match fs::metadata(store_path.join(&sub.raw_relpath)) {
|
||||
Ok(meta) => meta.len() as i64,
|
||||
Err(e) => {
|
||||
eprintln!("warn: skipping archived subtitle {relpath}: {e:#}");
|
||||
continue;
|
||||
}
|
||||
};
|
||||
let blob_id = database::upsert_blob(
|
||||
&tx,
|
||||
&BlobRecord {
|
||||
sha256,
|
||||
byte_size,
|
||||
mime_type: Some(sub.format.mime().to_string()),
|
||||
extension: Some(sub.format.extension().to_string()),
|
||||
raw_relpath: relpath.clone(),
|
||||
},
|
||||
)?;
|
||||
if database::entry_has_artifact_blob(&tx, entry_id, SUBTITLE_ARTIFACT_ROLE, blob_id)? {
|
||||
continue;
|
||||
}
|
||||
database::add_entry_artifact(
|
||||
&tx,
|
||||
&NewArtifact {
|
||||
entry_id,
|
||||
artifact_role: SUBTITLE_ARTIFACT_ROLE.to_string(),
|
||||
storage_area: "raw".to_string(),
|
||||
relpath,
|
||||
blob_id: Some(blob_id),
|
||||
logical_path: None,
|
||||
metadata_json: Some(metadata.to_string()),
|
||||
},
|
||||
)?;
|
||||
inserted += 1;
|
||||
}
|
||||
tx.commit()?;
|
||||
database::refresh_entry_cached_bytes(conn, entry_id)?;
|
||||
Ok(inserted)
|
||||
}
|
||||
|
||||
/// Number of the entry's `subtitle` artifacts that reduce to a non-empty
|
||||
/// transcript. Unreadable or unsupported files do not count.
|
||||
pub(crate) fn usable_subtitle_count(
|
||||
conn: &Connection,
|
||||
store_path: &Path,
|
||||
entry_id: i64,
|
||||
) -> Result<usize> {
|
||||
let artifacts = database::list_entry_artifacts_by_role(conn, entry_id, SUBTITLE_ARTIFACT_ROLE)?;
|
||||
Ok(artifacts
|
||||
.iter()
|
||||
.filter(|a| {
|
||||
let ext = Path::new(&a.relpath)
|
||||
.extension()
|
||||
.and_then(|e| e.to_str())
|
||||
.unwrap_or("");
|
||||
SubtitleFormat::detect(ext, a.mime_type.as_deref().unwrap_or("")).is_some()
|
||||
&& fs::read_to_string(store_path.join(&a.relpath))
|
||||
.map(|raw| !subtitle_to_transcript(&raw).is_empty())
|
||||
.unwrap_or(false)
|
||||
})
|
||||
.count())
|
||||
}
|
||||
|
||||
/// `original_language` of the entry's first `subtitle` artifact (id order)
|
||||
/// whose metadata records one.
|
||||
fn existing_original_language(conn: &Connection, entry_id: i64) -> Result<Option<String>> {
|
||||
let artifacts = database::list_entry_artifacts_by_role(conn, entry_id, SUBTITLE_ARTIFACT_ROLE)?;
|
||||
Ok(artifacts
|
||||
.iter()
|
||||
.find_map(|a| parse_subtitle_metadata(a.metadata_json.as_deref()).original_language))
|
||||
}
|
||||
|
||||
/// Downloads subtitles for an existing YouTube video entry from its original
|
||||
/// URL and registers them. Returns the number of rows added by this call plus
|
||||
/// the video's original language, taken from the metadata probe when it
|
||||
/// succeeds and otherwise from existing subtitle artifacts.
|
||||
///
|
||||
/// Returns `added: 0` (no yt-dlp call) for non-YouTube-video entries or a
|
||||
/// missing / non-http(s) canonical URL, and without fetching when the entry
|
||||
/// already has a usable subtitle (concurrency re-check). An unreachable video
|
||||
/// or any yt-dlp failure is logged and counts as zero subtitles; only DB/IO
|
||||
/// errors propagate.
|
||||
pub fn fetch_subtitles_for_entry(
|
||||
paths: &ArchivePaths,
|
||||
entry_uid: &str,
|
||||
cookie_rules: &[database::CookieRule],
|
||||
) -> Result<SubtitleFetchOutcome> {
|
||||
let conn = database::open_or_initialize(&paths.archive_path)?;
|
||||
let info = database::entry_source_info(&conn, entry_uid)?
|
||||
.ok_or_else(|| anyhow!("entry not found: {entry_uid}"))?;
|
||||
let existing_language = existing_original_language(&conn, info.entry_id)?;
|
||||
let nothing_added = || SubtitleFetchOutcome {
|
||||
added: 0,
|
||||
original_language: existing_language.clone(),
|
||||
};
|
||||
if info.source_kind != "youtube" || info.entity_kind != "video" {
|
||||
return Ok(nothing_added());
|
||||
}
|
||||
let Some(url) = info
|
||||
.canonical_url
|
||||
.filter(|u| u.starts_with("https://") || u.starts_with("http://"))
|
||||
else {
|
||||
return Ok(nothing_added());
|
||||
};
|
||||
|
||||
let store_path = &paths.store_path;
|
||||
if usable_subtitle_count(&conn, store_path, info.entry_id)? > 0 {
|
||||
return Ok(nothing_added());
|
||||
}
|
||||
|
||||
let cookies = capture::resolve_cookies_for_url(cookie_rules, &url);
|
||||
let timeout = crate::summarizer::summary_cli_timeout();
|
||||
let Some(metadata) = ytdlp::fetch_metadata_with_timeout(&url, &cookies, Some(timeout)) else {
|
||||
eprintln!("warn: subtitle fetch for {entry_uid}: video unreachable ({url})");
|
||||
return Ok(nothing_added());
|
||||
};
|
||||
// Before planning: a video without captions is exactly the case that
|
||||
// needs its language for a transcription fallback.
|
||||
let original_language = serde_json::from_str::<serde_json::Value>(&metadata)
|
||||
.ok()
|
||||
.and_then(|v| ytdlp::original_language_from_metadata(&v))
|
||||
.or_else(|| existing_language.clone());
|
||||
let Some(request) = ytdlp::plan_subtitle_request(Some(&metadata)) else {
|
||||
eprintln!("info: subtitle fetch for {entry_uid}: no subtitle tracks available ({url})");
|
||||
return Ok(SubtitleFetchOutcome {
|
||||
added: 0,
|
||||
original_language,
|
||||
});
|
||||
};
|
||||
|
||||
let stage_key = format!("subs-{}", Uuid::new_v4().simple());
|
||||
let stage_dir = store_path.join("temp").join(&stage_key);
|
||||
let staged = match ytdlp::download_subtitles(&url, store_path, &stage_key, &request, &cookies, timeout) {
|
||||
Ok(staged) => staged,
|
||||
Err(e) => {
|
||||
eprintln!("warn: subtitle fetch for {entry_uid} failed: {e:#}");
|
||||
let _ = fs::remove_dir_all(&stage_dir);
|
||||
return Ok(SubtitleFetchOutcome {
|
||||
added: 0,
|
||||
original_language,
|
||||
});
|
||||
}
|
||||
};
|
||||
let archived = archive_staged_subtitles(store_path, staged);
|
||||
let _ = fs::remove_dir_all(&stage_dir);
|
||||
|
||||
let added = register_subtitle_artifacts(
|
||||
&conn,
|
||||
store_path,
|
||||
info.entry_id,
|
||||
&archived,
|
||||
SUBTITLE_ORIGIN_SUMMARY_FETCH,
|
||||
)?;
|
||||
eprintln!("info: subtitle fetch for {entry_uid}: registered {added} subtitle artifact(s)");
|
||||
Ok(SubtitleFetchOutcome {
|
||||
added,
|
||||
original_language,
|
||||
})
|
||||
}
|
||||
|
||||
/// Any `<...>` markup: `<c>`, `<c.colorE5E5E5>`, `<00:00:01.000>`, `<v Speaker>`, `<i>`.
|
||||
static TAG_RE: LazyLock<Regex> =
|
||||
LazyLock::new(|| Regex::new(r"<[^>]*>").expect("valid tag regex"));
|
||||
|
||||
/// SRT ASS override blocks such as `{\an8}`.
|
||||
static ASS_OVERRIDE_RE: LazyLock<Regex> =
|
||||
LazyLock::new(|| Regex::new(r"\{\\[^}]*\}").expect("valid ASS override regex"));
|
||||
|
||||
/// Strips markup from one cue text line and normalizes whitespace.
|
||||
fn clean_cue_line(line: &str) -> String {
|
||||
let without_tags = TAG_RE.replace_all(line, "");
|
||||
let without_ass = ASS_OVERRIDE_RE.replace_all(&without_tags, "");
|
||||
// `&` last so `&lt;` decodes to `<`, not `<`.
|
||||
let decoded = without_ass
|
||||
.replace("<", "<")
|
||||
.replace(">", ">")
|
||||
.replace(""", "\"")
|
||||
.replace("'", "'")
|
||||
.replace(" ", " ")
|
||||
.replace("&", "&");
|
||||
decoded.split_whitespace().collect::<Vec<_>>().join(" ")
|
||||
}
|
||||
|
||||
/// Appends `line` unless it repeats one of the last two lines; a line that
|
||||
/// extends the previous one (rolling auto-captions) replaces it.
|
||||
fn push_deduped(out: &mut Vec<String>, line: String) {
|
||||
let recent = &out[out.len().saturating_sub(2)..];
|
||||
if recent.iter().any(|l| *l == line) {
|
||||
return;
|
||||
}
|
||||
if let Some(last) = out.last_mut() {
|
||||
if line.len() > last.len() && line.starts_with(last.as_str()) {
|
||||
*last = line;
|
||||
return;
|
||||
}
|
||||
}
|
||||
out.push(line);
|
||||
}
|
||||
|
||||
/// Text lines of one cue block (everything after its timing line).
|
||||
fn reduce_block(block: &[&str], out: &mut Vec<String>) {
|
||||
let Some(timing) = block.iter().position(|l| l.contains("-->")) else {
|
||||
return; // WEBVTT header, NOTE, STYLE, REGION, bare index
|
||||
};
|
||||
for line in &block[timing + 1..] {
|
||||
let cleaned = clean_cue_line(line);
|
||||
if !cleaned.is_empty() {
|
||||
push_deduped(out, cleaned);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Reduces a VTT or SRT document to plain transcript text, one line per
|
||||
/// caption line, with markup, timings and rolling-caption repeats removed.
|
||||
///
|
||||
/// Blocks split only on truly empty lines: YouTube auto-caption cues contain
|
||||
/// lines holding a single space, which belong to the cue.
|
||||
pub fn subtitle_to_transcript(raw: &str) -> String {
|
||||
let text = raw
|
||||
.strip_prefix('\u{feff}')
|
||||
.unwrap_or(raw)
|
||||
.replace("\r\n", "\n")
|
||||
.replace('\r', "\n");
|
||||
let mut out: Vec<String> = Vec::new();
|
||||
let mut block: Vec<&str> = Vec::new();
|
||||
for line in text.split('\n') {
|
||||
if line.is_empty() {
|
||||
reduce_block(&block, &mut out);
|
||||
block.clear();
|
||||
} else {
|
||||
block.push(line);
|
||||
}
|
||||
}
|
||||
reduce_block(&block, &mut out);
|
||||
out.join("\n")
|
||||
}
|
||||
|
||||
/// Parses a `subtitle` artifact's `metadata_json`. Missing or invalid metadata
|
||||
/// yields language `""` and `Unknown` kind.
|
||||
pub fn parse_subtitle_metadata(metadata_json: Option<&str>) -> SubtitleTrackMeta {
|
||||
let value: serde_json::Value = metadata_json
|
||||
.and_then(|json| serde_json::from_str(json).ok())
|
||||
.unwrap_or(serde_json::Value::Null);
|
||||
let text = |key: &str| {
|
||||
value
|
||||
.get(key)
|
||||
.and_then(|v| v.as_str())
|
||||
.map(str::trim)
|
||||
.filter(|s| !s.is_empty())
|
||||
.map(str::to_string)
|
||||
};
|
||||
SubtitleTrackMeta {
|
||||
language: text("language").unwrap_or_default(),
|
||||
kind: text("kind")
|
||||
.map(|k| SubtitleKind::parse(&k))
|
||||
.unwrap_or(SubtitleKind::Unknown),
|
||||
original_language: text("original_language"),
|
||||
}
|
||||
}
|
||||
|
||||
/// Preference rank of a subtitle track for summaries; lower is better.
|
||||
///
|
||||
/// 0 manual English, 1 manual original-language, 2 other manual,
|
||||
/// 3 transcribed (any language), 4 auto/unknown original-language,
|
||||
/// 5 auto/unknown English, 6 anything else.
|
||||
pub fn subtitle_track_rank(meta: &SubtitleTrackMeta) -> u8 {
|
||||
let base = language_base(&meta.language);
|
||||
let is_en = base == "en";
|
||||
let is_orig = meta.language.to_ascii_lowercase().ends_with("-orig")
|
||||
|| (!base.is_empty()
|
||||
&& meta
|
||||
.original_language
|
||||
.as_deref()
|
||||
.is_some_and(|orig| language_base(orig) == base));
|
||||
match meta.kind {
|
||||
SubtitleKind::Manual if is_en => 0,
|
||||
SubtitleKind::Manual if is_orig => 1,
|
||||
SubtitleKind::Manual => 2,
|
||||
SubtitleKind::Transcribed => 3,
|
||||
_ if is_orig => 4,
|
||||
_ if is_en => 5,
|
||||
_ => 6,
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn meta(language: &str, kind: SubtitleKind, original: Option<&str>) -> SubtitleTrackMeta {
|
||||
SubtitleTrackMeta {
|
||||
language: language.to_string(),
|
||||
kind,
|
||||
original_language: original.map(str::to_string),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn vtt_reduction_strips_header_timestamps_settings_and_tags() {
|
||||
let vtt = "WEBVTT\nKind: captions\nLanguage: en\n\nNOTE a comment\nspanning lines\n\nSTYLE\n::cue { color: red }\n\ncue-1\n00:00:01.000 --> 00:00:03.000 align:start position:0%\n<v Speaker>Hello <i>there</i></v>\n\n00:00:03.000 --> 00:00:05.000\n<c.colorE5E5E5>General</c> <00:00:03.500><c>Kenobi</c>\n";
|
||||
assert_eq!(subtitle_to_transcript(vtt), "Hello there\nGeneral Kenobi");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn vtt_reduction_collapses_rolling_auto_captions() {
|
||||
// Real-shaped YouTube auto-caption VTT: each cue repeats the previous
|
||||
// line, 10 ms "freeze" cues duplicate it, and lines holding a single
|
||||
// space sit inside cues (they must not split blocks).
|
||||
let vtt = "WEBVTT\nKind: captions\nLanguage: en\n\n\
|
||||
00:00:00.000 --> 00:00:02.030 align:start position:0%\n \nhello<00:00:00.320><c> world</c><00:00:00.640><c> this</c>\n\n\
|
||||
00:00:02.030 --> 00:00:02.040 align:start position:0%\nhello world this\n \n\n\
|
||||
00:00:02.040 --> 00:00:04.110 align:start position:0%\nhello world this\nis<00:00:02.360><c> a</c><00:00:02.600><c> test</c>\n\n\
|
||||
00:00:04.110 --> 00:00:04.120 align:start position:0%\nis a test\n \n\n\
|
||||
00:00:04.120 --> 00:00:06.000 align:start position:0%\nis a test\nof<00:00:04.500><c> captions</c>\n\n\
|
||||
00:00:06.000 --> 00:00:06.010 align:start position:0%\nof captions\n \n";
|
||||
assert_eq!(
|
||||
subtitle_to_transcript(vtt),
|
||||
"hello world this\nis a test\nof captions"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn vtt_reduction_extends_growing_lines() {
|
||||
let vtt = "WEBVTT\n\n00:00:00.000 --> 00:00:01.000\nhello\n\n00:00:01.000 --> 00:00:02.000\nhello world\n";
|
||||
assert_eq!(subtitle_to_transcript(vtt), "hello world");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn srt_reduction_strips_indices_italics_and_ass_overrides() {
|
||||
let srt = "1\n00:00:01,000 --> 00:00:02,000\n<i>Hello</i> there\n\n2\n00:00:02,500 --> 00:00:04,000\n{\\an8}Second line\n<b>continues</b> here\n\n3\n00:00:04,000 --> 00:00:05,000\n42\n";
|
||||
assert_eq!(
|
||||
subtitle_to_transcript(srt),
|
||||
"Hello there\nSecond line\ncontinues here\n42"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn reduction_handles_bom_crlf_and_entities() {
|
||||
let vtt = "\u{feff}WEBVTT\r\n\r\n00:00:01.000 --> 00:00:02.000\r\nTom & Jerry <3 "cheese" it's &lt;\r\n\r\n00:00:02.000 --> 00:00:03.000\rold mac line\r";
|
||||
assert_eq!(
|
||||
subtitle_to_transcript(vtt),
|
||||
"Tom & Jerry <3 \"cheese\" it's <\nold mac line"
|
||||
);
|
||||
assert_eq!(subtitle_to_transcript(""), "");
|
||||
assert_eq!(subtitle_to_transcript("WEBVTT\n\n"), "");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn track_rank_prefers_manual_english_then_manual_original_then_auto_original() {
|
||||
let de = Some("de");
|
||||
assert_eq!(subtitle_track_rank(&meta("en", SubtitleKind::Manual, de)), 0);
|
||||
assert_eq!(subtitle_track_rank(&meta("en-GB", SubtitleKind::Manual, de)), 0);
|
||||
assert_eq!(subtitle_track_rank(&meta("de", SubtitleKind::Manual, de)), 1);
|
||||
assert_eq!(subtitle_track_rank(&meta("fr", SubtitleKind::Manual, de)), 2);
|
||||
assert_eq!(subtitle_track_rank(&meta("de-orig", SubtitleKind::Auto, de)), 4);
|
||||
assert_eq!(subtitle_track_rank(&meta("de-orig", SubtitleKind::Unknown, None)), 4);
|
||||
assert_eq!(subtitle_track_rank(&meta("en", SubtitleKind::Auto, de)), 5);
|
||||
assert_eq!(subtitle_track_rank(&meta("fr", SubtitleKind::Auto, de)), 6);
|
||||
assert_eq!(subtitle_track_rank(&meta("", SubtitleKind::Unknown, None)), 6);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn track_rank_places_transcribed_below_manual_above_auto() {
|
||||
let de = Some("de");
|
||||
let transcribed = subtitle_track_rank(&meta("fr", SubtitleKind::Transcribed, de));
|
||||
assert_eq!(transcribed, 3);
|
||||
assert_eq!(subtitle_track_rank(&meta("en", SubtitleKind::Transcribed, None)), 3);
|
||||
assert!(subtitle_track_rank(&meta("fr", SubtitleKind::Manual, de)) < transcribed);
|
||||
assert!(subtitle_track_rank(&meta("de-orig", SubtitleKind::Auto, de)) > transcribed);
|
||||
assert!(subtitle_track_rank(&meta("en", SubtitleKind::Auto, de)) > transcribed);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_subtitle_metadata_defaults_and_round_trip() {
|
||||
assert_eq!(
|
||||
parse_subtitle_metadata(None),
|
||||
meta("", SubtitleKind::Unknown, None)
|
||||
);
|
||||
assert_eq!(
|
||||
parse_subtitle_metadata(Some("not json")),
|
||||
meta("", SubtitleKind::Unknown, None)
|
||||
);
|
||||
assert_eq!(
|
||||
parse_subtitle_metadata(Some(
|
||||
r#"{"language":"de-orig","kind":"auto","format":"vtt","original_language":"de","origin":"capture"}"#
|
||||
)),
|
||||
meta("de-orig", SubtitleKind::Auto, Some("de"))
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn subtitle_format_detects_by_extension_or_mime() {
|
||||
assert_eq!(SubtitleFormat::detect("vtt", ""), Some(SubtitleFormat::Vtt));
|
||||
assert_eq!(SubtitleFormat::detect(".SRT", ""), Some(SubtitleFormat::Srt));
|
||||
assert_eq!(
|
||||
SubtitleFormat::detect("", "text/vtt; charset=utf-8"),
|
||||
Some(SubtitleFormat::Vtt)
|
||||
);
|
||||
assert_eq!(
|
||||
SubtitleFormat::detect("txt", "application/x-subrip"),
|
||||
Some(SubtitleFormat::Srt)
|
||||
);
|
||||
assert_eq!(SubtitleFormat::detect("ttml", "application/ttml+xml"), None);
|
||||
}
|
||||
|
||||
fn archive_fixture(
|
||||
source_kind: &str,
|
||||
entity_kind: &str,
|
||||
canonical_url: Option<&str>,
|
||||
) -> (tempfile::TempDir, ArchivePaths, database::ArchivedEntry) {
|
||||
let temp = tempfile::tempdir().unwrap();
|
||||
let paths = crate::archive::initialize_archive(
|
||||
temp.path(),
|
||||
&temp.path().join("store"),
|
||||
"Test archive",
|
||||
false,
|
||||
)
|
||||
.unwrap();
|
||||
let conn = database::open_or_initialize(&paths.archive_path).unwrap();
|
||||
let user_id = database::ensure_default_user(&conn).unwrap();
|
||||
let run = database::create_archive_run(&conn, user_id, 1).unwrap();
|
||||
let source_id = database::upsert_source_identity(
|
||||
&conn,
|
||||
source_kind,
|
||||
entity_kind,
|
||||
Some("fixture-1"),
|
||||
canonical_url,
|
||||
canonical_url.unwrap_or("fixture:1"),
|
||||
)
|
||||
.unwrap();
|
||||
let entry = database::create_archived_entry(
|
||||
&conn,
|
||||
&database::NewEntry {
|
||||
source_identity_id: source_id,
|
||||
archive_run_id: run.id,
|
||||
parent_entry_id: None,
|
||||
root_entry_id: None,
|
||||
created_by_user_id: user_id,
|
||||
owned_by_user_id: user_id,
|
||||
source_kind: source_kind.to_string(),
|
||||
entity_kind: entity_kind.to_string(),
|
||||
title: None,
|
||||
visibility: "private".to_string(),
|
||||
representation_kind: entity_kind.to_string(),
|
||||
source_metadata_json: "{}".to_string(),
|
||||
display_metadata_json: None,
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
(temp, paths, entry)
|
||||
}
|
||||
|
||||
fn stage_vtt(store_path: &Path, name: &str, body: &str) -> StagedSubtitle {
|
||||
let dir = store_path.join("temp").join("stage");
|
||||
fs::create_dir_all(&dir).unwrap();
|
||||
let path = dir.join(name);
|
||||
fs::write(&path, body).unwrap();
|
||||
StagedSubtitle {
|
||||
path,
|
||||
language: "de-orig".to_string(),
|
||||
kind: SubtitleKind::Auto,
|
||||
format: "vtt".to_string(),
|
||||
original_language: Some("de".to_string()),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn register_subtitle_artifacts_dedups_same_blob() {
|
||||
let (_temp, paths, entry) =
|
||||
archive_fixture("youtube", "video", Some("https://www.youtube.com/watch?v=x"));
|
||||
let store_path = &paths.store_path;
|
||||
let body = "WEBVTT\n\n00:00:00.000 --> 00:00:01.000\nHallo Welt\n";
|
||||
let conn = database::open_or_initialize(&paths.archive_path).unwrap();
|
||||
|
||||
let first = archive_staged_subtitles(store_path, vec![stage_vtt(store_path, "a.de-orig.vtt", body)]);
|
||||
assert_eq!(first.len(), 1);
|
||||
assert!(store_path.join(&first[0].raw_relpath).is_file());
|
||||
assert_eq!(
|
||||
register_subtitle_artifacts(&conn, store_path, entry.id, &first, SUBTITLE_ORIGIN_CAPTURE)
|
||||
.unwrap(),
|
||||
1
|
||||
);
|
||||
|
||||
// Same bytes fetched again: raw move dedupes, registration skips.
|
||||
let second = archive_staged_subtitles(store_path, vec![stage_vtt(store_path, "b.de-orig.vtt", body)]);
|
||||
assert_eq!(second[0].raw_relpath, first[0].raw_relpath);
|
||||
assert_eq!(
|
||||
register_subtitle_artifacts(
|
||||
&conn,
|
||||
store_path,
|
||||
entry.id,
|
||||
&second,
|
||||
SUBTITLE_ORIGIN_SUMMARY_FETCH
|
||||
)
|
||||
.unwrap(),
|
||||
0
|
||||
);
|
||||
|
||||
let rows =
|
||||
database::list_entry_artifacts_by_role(&conn, entry.id, SUBTITLE_ARTIFACT_ROLE).unwrap();
|
||||
assert_eq!(rows.len(), 1);
|
||||
assert_eq!(rows[0].mime_type.as_deref(), Some("text/vtt"));
|
||||
assert_eq!(rows[0].relpath, first[0].raw_relpath.to_string_lossy());
|
||||
assert_eq!(
|
||||
parse_subtitle_metadata(rows[0].metadata_json.as_deref()),
|
||||
meta("de-orig", SubtitleKind::Auto, Some("de"))
|
||||
);
|
||||
let stored: serde_json::Value =
|
||||
serde_json::from_str(rows[0].metadata_json.as_deref().unwrap()).unwrap();
|
||||
assert_eq!(stored["format"], "vtt");
|
||||
assert_eq!(stored["origin"], SUBTITLE_ORIGIN_CAPTURE);
|
||||
|
||||
assert_eq!(usable_subtitle_count(&conn, store_path, entry.id).unwrap(), 1);
|
||||
assert_eq!(
|
||||
register_subtitle_artifacts(&conn, store_path, entry.id, &[], SUBTITLE_ORIGIN_CAPTURE)
|
||||
.unwrap(),
|
||||
0
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn fetch_subtitles_for_entry_skips_non_youtube_and_non_http_entries() {
|
||||
// Each of these returns before any yt-dlp process could be spawned.
|
||||
let (_t1, web_paths, web) = archive_fixture("web", "page", Some("https://example.com/"));
|
||||
assert_eq!(
|
||||
fetch_subtitles_for_entry(&web_paths, &web.entry_uid, &[]).unwrap(),
|
||||
SubtitleFetchOutcome::default()
|
||||
);
|
||||
|
||||
let (_t2, offline_paths, offline) =
|
||||
archive_fixture("youtube", "video", Some("youtube-test:offline"));
|
||||
assert_eq!(
|
||||
fetch_subtitles_for_entry(&offline_paths, &offline.entry_uid, &[]).unwrap(),
|
||||
SubtitleFetchOutcome::default()
|
||||
);
|
||||
|
||||
let (_t3, no_url_paths, no_url) = archive_fixture("youtube", "video", None);
|
||||
assert_eq!(
|
||||
fetch_subtitles_for_entry(&no_url_paths, &no_url.entry_uid, &[]).unwrap(),
|
||||
SubtitleFetchOutcome::default()
|
||||
);
|
||||
|
||||
assert!(fetch_subtitles_for_entry(&web_paths, "entry_missing", &[]).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn fetch_outcome_reports_original_language_from_existing_artifacts() {
|
||||
let (_temp, paths, entry) =
|
||||
archive_fixture("youtube", "video", Some("youtube-test:offline"));
|
||||
let store_path = &paths.store_path;
|
||||
let conn = database::open_or_initialize(&paths.archive_path).unwrap();
|
||||
// An unusable (empty) track still carries the original language.
|
||||
let archived =
|
||||
archive_staged_subtitles(store_path, vec![stage_vtt(store_path, "e.de-orig.vtt", "WEBVTT\n")]);
|
||||
register_subtitle_artifacts(&conn, store_path, entry.id, &archived, SUBTITLE_ORIGIN_CAPTURE)
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
fetch_subtitles_for_entry(&paths, &entry.entry_uid, &[]).unwrap(),
|
||||
SubtitleFetchOutcome {
|
||||
added: 0,
|
||||
original_language: Some("de".to_string()),
|
||||
}
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn register_transcript_artifact_writes_engine_metadata_and_dedups() {
|
||||
let (_temp, paths, entry) =
|
||||
archive_fixture("youtube", "video", Some("https://www.youtube.com/watch?v=x"));
|
||||
let store_path = &paths.store_path;
|
||||
let conn = database::open_or_initialize(&paths.archive_path).unwrap();
|
||||
let body = "WEBVTT\n\n00:00:00.000 --> 00:00:02.000\nhello world\n";
|
||||
let staged = |name: &str| {
|
||||
let mut s = stage_vtt(store_path, name, body);
|
||||
s.language = "en".to_string();
|
||||
s.kind = SubtitleKind::Transcribed;
|
||||
s.original_language = None;
|
||||
s
|
||||
};
|
||||
|
||||
let first = archive_staged_subtitles(store_path, vec![staged("t1.vtt")]);
|
||||
assert_eq!(first.len(), 1);
|
||||
assert_eq!(
|
||||
register_transcript_artifact(
|
||||
&conn,
|
||||
store_path,
|
||||
entry.id,
|
||||
&first[0],
|
||||
"whisper",
|
||||
"/models/ggml-tiny.bin"
|
||||
)
|
||||
.unwrap(),
|
||||
1
|
||||
);
|
||||
let second = archive_staged_subtitles(store_path, vec![staged("t2.vtt")]);
|
||||
assert_eq!(
|
||||
register_transcript_artifact(
|
||||
&conn,
|
||||
store_path,
|
||||
entry.id,
|
||||
&second[0],
|
||||
"whisper",
|
||||
"/models/ggml-tiny.bin"
|
||||
)
|
||||
.unwrap(),
|
||||
0
|
||||
);
|
||||
|
||||
let rows =
|
||||
database::list_entry_artifacts_by_role(&conn, entry.id, SUBTITLE_ARTIFACT_ROLE).unwrap();
|
||||
assert_eq!(rows.len(), 1);
|
||||
assert_eq!(rows[0].mime_type.as_deref(), Some("text/vtt"));
|
||||
let stored: serde_json::Value =
|
||||
serde_json::from_str(rows[0].metadata_json.as_deref().unwrap()).unwrap();
|
||||
assert_eq!(stored["kind"], "transcribed");
|
||||
assert_eq!(stored["origin"], SUBTITLE_ORIGIN_TRANSCRIPTION);
|
||||
assert_eq!(stored["engine"], "whisper");
|
||||
assert_eq!(stored["model"], "ggml-tiny.bin");
|
||||
assert_eq!(stored["language"], "en");
|
||||
assert_eq!(
|
||||
parse_subtitle_metadata(rows[0].metadata_json.as_deref()).kind,
|
||||
SubtitleKind::Transcribed
|
||||
);
|
||||
|
||||
assert_eq!(sanitize_model_name("nvidia/parakeet-tdt-0.6b-v3"), "nvidia/parakeet-tdt-0.6b-v3");
|
||||
assert_eq!(sanitize_model_name("phonon-2"), "phonon-2");
|
||||
assert_eq!(sanitize_model_name("C:\\models\\x.bin"), "x.bin");
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load diff
|
|
@ -1,650 +0,0 @@
|
|||
//! On-demand titles for archived X threads.
|
||||
//!
|
||||
//! A user asks for a title from the entry rail; the ordered thread text is sent
|
||||
//! to the selected summary provider with a cheap per-provider model and the
|
||||
//! result is saved as `Thread about <topic> — @author` (just `Thread about
|
||||
//! <topic>` when the author is unknown).
|
||||
//!
|
||||
//! The title model never inherits the summary model (`ARCHIVR_*_MODEL`). It is
|
||||
//! resolved as: the admin's instance setting (passed in by the caller; core
|
||||
//! never reads the auth DB) > `ARCHIVR_ANTHROPIC_TITLE_MODEL` /
|
||||
//! `ARCHIVR_OPENAI_TITLE_MODEL` / `ARCHIVR_CLAUDE_TITLE_MODEL` /
|
||||
//! `ARCHIVR_CODEX_TITLE_MODEL` > a built-in small default. Endpoint, key, CLI
|
||||
//! path and timeout still come from `summarizer::provider_from_env`.
|
||||
//!
|
||||
//! The model returns only the topic phrase; the server builds the rest so the
|
||||
//! title format (prefix and author suffix) is guaranteed regardless of output.
|
||||
|
||||
use anyhow::{Context, Result, bail};
|
||||
use rusqlite::OptionalExtension;
|
||||
use std::path::Path;
|
||||
|
||||
use crate::archive::ArchivePaths;
|
||||
use crate::database;
|
||||
use crate::env_config::optional_env;
|
||||
use crate::summarizer::{self, ProviderConfig};
|
||||
|
||||
pub const TITLE_MAX_TOKENS: u32 = 64;
|
||||
const MAX_TITLE_INPUT_CHARS: usize = 8_000;
|
||||
const MAX_TOPIC_WORDS: usize = 10;
|
||||
const MAX_TOPIC_CHARS: usize = 80;
|
||||
|
||||
const TITLE_SYSTEM_PROMPT: &str = "You name archived X (Twitter) threads for a personal archive index. Reply with ONLY a short topic phrase of 3 to 8 words that completes the sentence 'Thread about …' (for example: migrating a home server to NixOS). Plain text on one line: no quotes, no markdown, no hashtags, no emoji, no @mentions, no trailing punctuation, and do not repeat the words 'Thread about'.";
|
||||
|
||||
const QUOTE_CHARS: &[char] = &['"', '\'', '`', '“', '”', '‘', '’', '«', '»', '*', '_', '#'];
|
||||
const TRAILING_PUNCT: &[char] = &['.', ',', ';', ':', '!', '?', '…'];
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct ThreadTitleInput {
|
||||
pub entry_uid: String,
|
||||
/// Empty when no status JSON names the author.
|
||||
pub author: String,
|
||||
pub content: String,
|
||||
}
|
||||
|
||||
pub fn title_model_env(kind: &str) -> Option<&'static str> {
|
||||
match kind {
|
||||
"anthropic_http" => Some("ARCHIVR_ANTHROPIC_TITLE_MODEL"),
|
||||
"openai_compatible" => Some("ARCHIVR_OPENAI_TITLE_MODEL"),
|
||||
"claude_cli" => Some("ARCHIVR_CLAUDE_TITLE_MODEL"),
|
||||
"codex_cli" => Some("ARCHIVR_CODEX_TITLE_MODEL"),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
pub fn default_title_model(kind: &str) -> Option<&'static str> {
|
||||
match kind {
|
||||
"anthropic_http" => Some("claude-haiku-4-5"),
|
||||
"openai_compatible" => Some("gpt-4o-mini"),
|
||||
"claude_cli" => Some("haiku"),
|
||||
"codex_cli" => Some("gpt-6-luna"),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
pub fn with_title_model(cfg: ProviderConfig, model: String) -> ProviderConfig {
|
||||
match cfg {
|
||||
ProviderConfig::AnthropicHttp(mut c) => {
|
||||
c.model = model;
|
||||
ProviderConfig::AnthropicHttp(c)
|
||||
}
|
||||
ProviderConfig::OpenAiCompatible(mut c) => {
|
||||
c.model = model;
|
||||
ProviderConfig::OpenAiCompatible(c)
|
||||
}
|
||||
ProviderConfig::ClaudeCli(mut c) => {
|
||||
c.model = Some(model);
|
||||
ProviderConfig::ClaudeCli(c)
|
||||
}
|
||||
ProviderConfig::CodexCli(mut c) => {
|
||||
c.model = Some(model);
|
||||
ProviderConfig::CodexCli(c)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Where an effective title model came from.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum TitleModelSource {
|
||||
Instance,
|
||||
Env,
|
||||
Default,
|
||||
}
|
||||
|
||||
impl TitleModelSource {
|
||||
pub fn as_str(self) -> &'static str {
|
||||
match self {
|
||||
Self::Instance => "instance",
|
||||
Self::Env => "env",
|
||||
Self::Default => "default",
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Effective title model for `kind`: non-empty trimmed `instance_override` >
|
||||
/// non-empty title env var > built-in default. `None` for unknown kinds.
|
||||
pub fn resolve_title_model(
|
||||
kind: &str,
|
||||
instance_override: Option<&str>,
|
||||
) -> Option<(String, TitleModelSource)> {
|
||||
let (var, default) = (title_model_env(kind)?, default_title_model(kind)?);
|
||||
if let Some(m) = instance_override.map(str::trim).filter(|m| !m.is_empty()) {
|
||||
return Some((m.to_string(), TitleModelSource::Instance));
|
||||
}
|
||||
if let Some(m) = optional_env(var).map(|m| m.trim().to_string()).filter(|m| !m.is_empty()) {
|
||||
return Some((m, TitleModelSource::Env));
|
||||
}
|
||||
Some((default.to_string(), TitleModelSource::Default))
|
||||
}
|
||||
|
||||
/// Provider config for title generation: transport settings from the summary
|
||||
/// env, model from [`resolve_title_model`].
|
||||
pub fn title_provider_from_env(
|
||||
kind: &str,
|
||||
instance_override: Option<&str>,
|
||||
) -> Result<ProviderConfig> {
|
||||
// Validates `kind` and keeps the summary path's missing-key messages.
|
||||
let cfg = summarizer::provider_from_env(kind)?;
|
||||
let Some((model, _)) = resolve_title_model(kind, instance_override) else {
|
||||
bail!("unknown summary provider: {kind}");
|
||||
};
|
||||
Ok(with_title_model(cfg, model))
|
||||
}
|
||||
|
||||
/// Expected, user-facing failure of [`load_thread_title_input`] (entry is not a
|
||||
/// thread, or has no archived text). Anything else is an internal error.
|
||||
#[derive(Debug)]
|
||||
pub struct ThreadTitleUserError(pub String);
|
||||
|
||||
impl std::fmt::Display for ThreadTitleUserError {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
f.write_str(&self.0)
|
||||
}
|
||||
}
|
||||
|
||||
impl std::error::Error for ThreadTitleUserError {}
|
||||
|
||||
/// The [`ThreadTitleUserError`] message carried by `error`, if any.
|
||||
pub fn thread_title_user_message(error: &anyhow::Error) -> Option<String> {
|
||||
error.downcast_ref::<ThreadTitleUserError>().map(|m| m.0.clone())
|
||||
}
|
||||
|
||||
/// Loads the thread text and author. `Ok(None)` means the entry does not exist.
|
||||
pub fn load_thread_title_input(
|
||||
paths: &ArchivePaths,
|
||||
entry_uid: &str,
|
||||
) -> Result<Option<ThreadTitleInput>> {
|
||||
let conn = database::open_or_initialize(&paths.archive_path)?;
|
||||
let Some((id, entity_kind, source_metadata_json)) = conn
|
||||
.query_row(
|
||||
"SELECT id, entity_kind, source_metadata_json FROM archived_entries WHERE entry_uid = ?1",
|
||||
[entry_uid],
|
||||
|row| Ok((row.get::<_, i64>(0)?, row.get::<_, String>(1)?, row.get::<_, String>(2)?)),
|
||||
)
|
||||
.optional()?
|
||||
else {
|
||||
return Ok(None);
|
||||
};
|
||||
if entity_kind != "tweet_thread" {
|
||||
return Err(anyhow::Error::new(ThreadTitleUserError(format!(
|
||||
"entry is '{entity_kind}', not an X thread; titles can only be generated for threads"
|
||||
))));
|
||||
}
|
||||
|
||||
let content = summarizer::artifact_text_content(&conn, &paths.store_path, id, &entity_kind)
|
||||
.map_err(|e| {
|
||||
if summarizer::is_unsupported_summary_content_error(&e) {
|
||||
anyhow::Error::new(ThreadTitleUserError(
|
||||
"this thread has no archived text to generate a title from".to_string(),
|
||||
))
|
||||
} else {
|
||||
e
|
||||
}
|
||||
})?;
|
||||
let content: String = content.chars().take(MAX_TITLE_INPUT_CHARS).collect();
|
||||
|
||||
let root_tweet_id = serde_json::from_str::<serde_json::Value>(&source_metadata_json)
|
||||
.ok()
|
||||
.and_then(|v| v["tweet_id"].as_str().map(str::to_string));
|
||||
let author = thread_author(
|
||||
&conn,
|
||||
&paths.store_path,
|
||||
id,
|
||||
root_tweet_id.as_deref(),
|
||||
entry_uid,
|
||||
)?;
|
||||
|
||||
Ok(Some(ThreadTitleInput {
|
||||
entry_uid: entry_uid.to_string(),
|
||||
author,
|
||||
content,
|
||||
}))
|
||||
}
|
||||
|
||||
/// Author of the root status (`tweet-<source_metadata.tweet_id>.json`, as in
|
||||
/// capture's `Thread by @…`), else of the first readable status JSON.
|
||||
fn thread_author(
|
||||
conn: &rusqlite::Connection,
|
||||
store_path: &Path,
|
||||
entry_id: i64,
|
||||
root_tweet_id: Option<&str>,
|
||||
entry_uid: &str,
|
||||
) -> Result<String> {
|
||||
let root_file = root_tweet_id.map(|id| format!("tweet-{id}.json"));
|
||||
for role in ["raw_tweet_json", "primary_media"] {
|
||||
let mut artifacts: Vec<_> = database::list_entry_artifacts_by_role(conn, entry_id, role)?
|
||||
.into_iter()
|
||||
.filter(|a| a.relpath.ends_with(".json"))
|
||||
.collect();
|
||||
if artifacts.is_empty() {
|
||||
continue;
|
||||
}
|
||||
// Root status first; the rest keep insertion order (stable sort).
|
||||
if let Some(root_file) = root_file.as_deref() {
|
||||
artifacts.sort_by_key(|a| {
|
||||
!(Path::new(&a.relpath).file_name().and_then(|n| n.to_str()) == Some(root_file))
|
||||
});
|
||||
}
|
||||
for artifact in &artifacts {
|
||||
let abs = store_path.join(&artifact.relpath);
|
||||
let parsed = std::fs::read_to_string(&abs)
|
||||
.with_context(|| format!("failed to read {}", abs.display()))
|
||||
.and_then(|raw| {
|
||||
serde_json::from_str::<serde_json::Value>(&raw)
|
||||
.with_context(|| format!("{} is not valid JSON", abs.display()))
|
||||
});
|
||||
let json = match parsed {
|
||||
Ok(json) => json,
|
||||
Err(e) => {
|
||||
eprintln!("warn: thread title {entry_uid}: {e:#}");
|
||||
continue;
|
||||
}
|
||||
};
|
||||
let name = json["author"]["screen_name"]
|
||||
.as_str()
|
||||
.map(|s| s.trim().trim_start_matches('@').trim())
|
||||
.filter(|s| !s.is_empty());
|
||||
if let Some(name) = name {
|
||||
return Ok(name.to_string());
|
||||
}
|
||||
}
|
||||
// Only the first role that has JSON artifacts is consulted, matching
|
||||
// `artifact_text_content`'s legacy `primary_media` fallback.
|
||||
break;
|
||||
}
|
||||
Ok(String::new())
|
||||
}
|
||||
|
||||
pub fn build_title_user_prompt(input: &ThreadTitleInput) -> String {
|
||||
if input.author.is_empty() {
|
||||
format!("Thread:\n{}\n", input.content)
|
||||
} else {
|
||||
format!("Author: @{}\n\nThread:\n{}\n", input.author, input.content)
|
||||
}
|
||||
}
|
||||
|
||||
fn strip_prefix_ci<'a>(s: &'a str, prefix: &str) -> Option<&'a str> {
|
||||
let head = s.get(..prefix.len())?;
|
||||
head.eq_ignore_ascii_case(prefix).then(|| &s[prefix.len()..])
|
||||
}
|
||||
|
||||
fn trim_trailing_punct(s: &str) -> &str {
|
||||
s.trim_end_matches(|c: char| TRAILING_PUNCT.contains(&c) || QUOTE_CHARS.contains(&c))
|
||||
.trim_end()
|
||||
}
|
||||
|
||||
/// Reduces a model reply to a single short topic phrase.
|
||||
pub fn sanitize_topic(raw: &str) -> Result<String> {
|
||||
let line = raw
|
||||
.lines()
|
||||
.filter(|l| !l.trim().starts_with("```"))
|
||||
.map(str::trim)
|
||||
.find(|l| !l.is_empty())
|
||||
.unwrap_or("");
|
||||
let line: String = line.chars().filter(|c| !c.is_control()).collect();
|
||||
|
||||
let mut s: &str = line.trim();
|
||||
for prefix in ["title:", "topic:"] {
|
||||
if let Some(rest) = strip_prefix_ci(s, prefix) {
|
||||
s = rest;
|
||||
break;
|
||||
}
|
||||
}
|
||||
loop {
|
||||
let before = s;
|
||||
s = s.trim().trim_matches(|c: char| QUOTE_CHARS.contains(&c));
|
||||
if let Some(rest) = strip_prefix_ci(s, "thread about ") {
|
||||
s = rest;
|
||||
}
|
||||
if let Some(rest) = strip_prefix_ci(s, "about ") {
|
||||
s = rest;
|
||||
}
|
||||
if s == before {
|
||||
break;
|
||||
}
|
||||
}
|
||||
for sep in [" — @", " – @", " - @"] {
|
||||
if let Some(idx) = s.find(sep) {
|
||||
s = &s[..idx];
|
||||
}
|
||||
}
|
||||
|
||||
let collapsed = s.split_whitespace().collect::<Vec<_>>().join(" ");
|
||||
let trimmed = trim_trailing_punct(&collapsed);
|
||||
let mut topic = trimmed
|
||||
.split(' ')
|
||||
.filter(|w| !w.is_empty())
|
||||
.take(MAX_TOPIC_WORDS)
|
||||
.collect::<Vec<_>>()
|
||||
.join(" ");
|
||||
|
||||
if topic.chars().count() > MAX_TOPIC_CHARS {
|
||||
let head: String = topic.chars().take(MAX_TOPIC_CHARS).collect();
|
||||
// `head` is MAX_TOPIC_CHARS chars and the next char exists, so a space
|
||||
// at the cut point is preserved by checking the following char too.
|
||||
let next_is_space = topic.chars().nth(MAX_TOPIC_CHARS) == Some(' ');
|
||||
let cut = if next_is_space {
|
||||
head.as_str()
|
||||
} else {
|
||||
match head.rfind(' ') {
|
||||
Some(idx) => &head[..idx],
|
||||
None => head.as_str(),
|
||||
}
|
||||
};
|
||||
topic = trim_trailing_punct(cut).to_string();
|
||||
}
|
||||
|
||||
if topic.is_empty() {
|
||||
bail!("title provider returned no usable title");
|
||||
}
|
||||
Ok(topic)
|
||||
}
|
||||
|
||||
/// `author` is empty when unknown; the ` — @…` suffix is then omitted.
|
||||
pub fn format_thread_title(topic: &str, author: &str) -> String {
|
||||
if author.is_empty() {
|
||||
format!("Thread about {topic}")
|
||||
} else {
|
||||
format!("Thread about {topic} — @{author}")
|
||||
}
|
||||
}
|
||||
|
||||
fn short(s: &str) -> String {
|
||||
s.chars().take(200).collect()
|
||||
}
|
||||
|
||||
/// Asks the provider for a topic and builds the final title.
|
||||
pub fn generate_thread_title(cfg: &ProviderConfig, input: &ThreadTitleInput) -> Result<String> {
|
||||
let out = summarizer::complete_plain(
|
||||
cfg,
|
||||
TITLE_SYSTEM_PROMPT,
|
||||
&build_title_user_prompt(input),
|
||||
TITLE_MAX_TOKENS,
|
||||
)
|
||||
.context("title generation failed")?;
|
||||
let topic =
|
||||
sanitize_topic(&out.text).with_context(|| format!("raw reply: {}", short(&out.text)))?;
|
||||
Ok(format_thread_title(&topic, &input.author))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::summarizer::{CliProviderConfig, HttpProviderConfig};
|
||||
|
||||
#[test]
|
||||
fn sanitize_topic_cleans_model_replies() {
|
||||
let ok = |raw: &str| sanitize_topic(raw).unwrap();
|
||||
assert_eq!(ok("\"Rust async runtimes compared.\""), "Rust async runtimes compared");
|
||||
assert_eq!(ok("Thread about NixOS on a Pi"), "NixOS on a Pi");
|
||||
assert_eq!(ok("```\nTitle: **Home lab networking**\n```"), "Home lab networking");
|
||||
assert_eq!(ok("\nFirst line topic\nSecond line"), "First line topic");
|
||||
assert_eq!(ok("Foo bar — @alice"), "Foo bar");
|
||||
let twenty = (1..=20).map(|i| format!("w{i}")).collect::<Vec<_>>().join(" ");
|
||||
assert_eq!(ok(&twenty), "w1 w2 w3 w4 w5 w6 w7 w8 w9 w10");
|
||||
let long_word = "x".repeat(120);
|
||||
assert!(ok(&long_word).chars().count() <= MAX_TOPIC_CHARS);
|
||||
let long_words = vec!["abcdefghijk"; 10].join(" ");
|
||||
let cut = ok(&long_words);
|
||||
assert!(cut.chars().count() <= MAX_TOPIC_CHARS);
|
||||
assert!(!cut.ends_with(' '));
|
||||
assert!(sanitize_topic(" ").is_err());
|
||||
assert!(sanitize_topic("\"\"").is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn format_thread_title_uses_em_dash_suffix() {
|
||||
assert_eq!(format_thread_title("x y", "bob"), "Thread about x y — @bob");
|
||||
assert_eq!(format_thread_title("x y", ""), "Thread about x y");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn title_models_cover_all_provider_kinds() {
|
||||
for kind in summarizer::PROVIDER_KINDS {
|
||||
assert!(title_model_env(kind).is_some(), "{kind}");
|
||||
assert!(default_title_model(kind).is_some(), "{kind}");
|
||||
}
|
||||
assert_eq!(title_model_env("claude_cli"), Some("ARCHIVR_CLAUDE_TITLE_MODEL"));
|
||||
assert_eq!(default_title_model("codex_cli"), Some("gpt-6-luna"));
|
||||
assert_eq!(default_title_model("anthropic_http"), Some("claude-haiku-4-5"));
|
||||
assert_eq!(title_model_env("gemini"), None);
|
||||
assert_eq!(default_title_model("gemini"), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn with_title_model_sets_model_on_every_variant() {
|
||||
let http = HttpProviderConfig {
|
||||
endpoint: "https://example.invalid".into(),
|
||||
api_key: "k".into(),
|
||||
model: "big".into(),
|
||||
timeout_secs: 1,
|
||||
};
|
||||
let cli = CliProviderConfig {
|
||||
executable: "claude".into(),
|
||||
model: None,
|
||||
timeout_secs: 1,
|
||||
};
|
||||
let m = || "small".to_string();
|
||||
match with_title_model(ProviderConfig::AnthropicHttp(http.clone()), m()) {
|
||||
ProviderConfig::AnthropicHttp(c) => assert_eq!(c.model, "small"),
|
||||
other => panic!("{other:?}"),
|
||||
}
|
||||
match with_title_model(ProviderConfig::OpenAiCompatible(http), m()) {
|
||||
ProviderConfig::OpenAiCompatible(c) => assert_eq!(c.model, "small"),
|
||||
other => panic!("{other:?}"),
|
||||
}
|
||||
match with_title_model(ProviderConfig::ClaudeCli(cli.clone()), m()) {
|
||||
ProviderConfig::ClaudeCli(c) => assert_eq!(c.model.as_deref(), Some("small")),
|
||||
other => panic!("{other:?}"),
|
||||
}
|
||||
match with_title_model(ProviderConfig::CodexCli(cli), m()) {
|
||||
ProviderConfig::CodexCli(c) => assert_eq!(c.model.as_deref(), Some("small")),
|
||||
other => panic!("{other:?}"),
|
||||
}
|
||||
}
|
||||
|
||||
/// Archive with one entry of `entity_kind` and the given raw tweet JSON files.
|
||||
fn fixture(
|
||||
entity_kind: &str,
|
||||
tweets: &[(&str, serde_json::Value)],
|
||||
) -> (tempfile::TempDir, ArchivePaths, database::ArchivedEntry) {
|
||||
let temp = tempfile::tempdir().unwrap();
|
||||
let paths = crate::archive::initialize_archive(
|
||||
temp.path(),
|
||||
&temp.path().join("store"),
|
||||
"Test archive",
|
||||
false,
|
||||
)
|
||||
.unwrap();
|
||||
let conn = database::open_or_initialize(&paths.archive_path).unwrap();
|
||||
let user_id = database::ensure_default_user(&conn).unwrap();
|
||||
let run = database::create_archive_run(&conn, user_id, 1).unwrap();
|
||||
let source_id = database::upsert_source_identity(
|
||||
&conn,
|
||||
"x",
|
||||
entity_kind,
|
||||
Some("9001"),
|
||||
Some("https://x.com/alice/status/9001"),
|
||||
"x:thread:9001",
|
||||
)
|
||||
.unwrap();
|
||||
let entry = database::create_archived_entry(
|
||||
&conn,
|
||||
&database::NewEntry {
|
||||
source_identity_id: source_id,
|
||||
archive_run_id: run.id,
|
||||
parent_entry_id: None,
|
||||
root_entry_id: None,
|
||||
created_by_user_id: user_id,
|
||||
owned_by_user_id: user_id,
|
||||
source_kind: "x".to_string(),
|
||||
entity_kind: entity_kind.to_string(),
|
||||
title: Some("Thread by @alice".to_string()),
|
||||
visibility: "private".to_string(),
|
||||
representation_kind: entity_kind.to_string(),
|
||||
source_metadata_json: r#"{"tweet_id":"9001"}"#.to_string(),
|
||||
display_metadata_json: None,
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
std::fs::create_dir_all(paths.store_path.join("raw_tweets")).unwrap();
|
||||
for (relpath, body) in tweets {
|
||||
std::fs::write(paths.store_path.join(relpath), body.to_string()).unwrap();
|
||||
database::add_entry_artifact(
|
||||
&conn,
|
||||
&database::NewArtifact {
|
||||
entry_id: entry.id,
|
||||
artifact_role: "raw_tweet_json".to_string(),
|
||||
storage_area: "raw_tweets".to_string(),
|
||||
relpath: relpath.to_string(),
|
||||
blob_id: None,
|
||||
logical_path: None,
|
||||
metadata_json: None,
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
}
|
||||
(temp, paths, entry)
|
||||
}
|
||||
|
||||
fn alice_thread() -> Vec<(&'static str, serde_json::Value)> {
|
||||
vec![
|
||||
(
|
||||
"raw_tweets/tweet-9001.json",
|
||||
serde_json::json!({
|
||||
"full_text": "1/ Comparing Rust async runtimes.",
|
||||
"author": { "screen_name": "@alice" }
|
||||
}),
|
||||
),
|
||||
(
|
||||
"raw_tweets/tweet-9002.json",
|
||||
serde_json::json!({
|
||||
"full_text": "2/ Tokio wins on ecosystem.",
|
||||
"author": { "screen_name": "alice" }
|
||||
}),
|
||||
),
|
||||
]
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn load_thread_title_input_reads_thread_text_and_author() {
|
||||
let (_temp, paths, entry) = fixture("tweet_thread", &alice_thread());
|
||||
let input = load_thread_title_input(&paths, &entry.entry_uid)
|
||||
.unwrap()
|
||||
.unwrap();
|
||||
assert_eq!(input.entry_uid, entry.entry_uid);
|
||||
assert_eq!(input.author, "alice");
|
||||
assert!(
|
||||
input
|
||||
.content
|
||||
.contains("1/ Comparing Rust async runtimes.\n\n---\n\n2/ Tokio wins on ecosystem."),
|
||||
"{}",
|
||||
input.content
|
||||
);
|
||||
assert!(!input.content.contains("Thread by @alice"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn load_thread_title_input_prefers_root_status_author() {
|
||||
let tweets = vec![
|
||||
(
|
||||
"raw_tweets/tweet-8000.json",
|
||||
serde_json::json!({ "full_text": "quoted", "author": { "screen_name": "bob" } }),
|
||||
),
|
||||
(
|
||||
"raw_tweets/tweet-9001.json",
|
||||
serde_json::json!({ "full_text": "root", "author": { "screen_name": "alice" } }),
|
||||
),
|
||||
];
|
||||
let (_temp, paths, entry) = fixture("tweet_thread", &tweets);
|
||||
let input = load_thread_title_input(&paths, &entry.entry_uid)
|
||||
.unwrap()
|
||||
.unwrap();
|
||||
assert_eq!(input.author, "alice");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn load_thread_title_input_rejects_non_threads_and_empty_threads() {
|
||||
let (_temp, paths, entry) = fixture("page", &[]);
|
||||
let err = load_thread_title_input(&paths, &entry.entry_uid).unwrap_err();
|
||||
assert!(format!("{err:#}").contains("not an X thread"), "{err:#}");
|
||||
assert!(thread_title_user_message(&err).is_some());
|
||||
assert!(load_thread_title_input(&paths, "no-such-uid").unwrap().is_none());
|
||||
|
||||
let empty = vec![(
|
||||
"raw_tweets/tweet-9001.json",
|
||||
serde_json::json!({ "full_text": "", "author": { "screen_name": "alice" } }),
|
||||
)];
|
||||
let (_temp2, paths2, entry2) = fixture("tweet_thread", &empty);
|
||||
let err = load_thread_title_input(&paths2, &entry2.entry_uid).unwrap_err();
|
||||
assert!(format!("{err:#}").contains("no archived text"), "{err:#}");
|
||||
assert!(thread_title_user_message(&err).is_some());
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
#[test]
|
||||
fn generate_thread_title_runs_claude_cli_with_title_model() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let script = dir.path().join("fake-claude");
|
||||
crate::downloader::write_script(
|
||||
&script,
|
||||
"#!/bin/sh\ncat >/dev/null\ncase \" $* \" in *\" --model haiku \"*) ;; *) echo \"bad args: $*\" >&2; exit 3;; esac\nprintf '%s\\n' '\"Rust async runtimes compared.\"'\n",
|
||||
);
|
||||
let cfg = ProviderConfig::ClaudeCli(CliProviderConfig {
|
||||
executable: script,
|
||||
model: Some("haiku".into()),
|
||||
timeout_secs: 30,
|
||||
});
|
||||
let input = ThreadTitleInput {
|
||||
entry_uid: "uid".into(),
|
||||
author: "alice".into(),
|
||||
content: "1/ Comparing Rust async runtimes.".into(),
|
||||
};
|
||||
// Parallel tests forking while the script fd was open can briefly make
|
||||
// exec fail with ETXTBSY (rust-lang/rust#114554); retry that case only.
|
||||
let mut attempt = 0;
|
||||
let title = loop {
|
||||
match generate_thread_title(&cfg, &input) {
|
||||
Err(e) if attempt < 20 && format!("{e:#}").contains("Text file busy") => {
|
||||
attempt += 1;
|
||||
std::thread::sleep(std::time::Duration::from_millis(50));
|
||||
}
|
||||
other => break other.unwrap(),
|
||||
}
|
||||
};
|
||||
assert_eq!(title, "Thread about Rust async runtimes compared — @alice");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn resolve_title_model_prefers_instance_then_env_then_default() {
|
||||
// Only this test touches the codex title env var; restored below.
|
||||
let var = "ARCHIVR_CODEX_TITLE_MODEL";
|
||||
let previous = std::env::var_os(var);
|
||||
unsafe { std::env::remove_var(var) };
|
||||
assert_eq!(
|
||||
resolve_title_model("codex_cli", None),
|
||||
Some(("gpt-6-luna".into(), TitleModelSource::Default))
|
||||
);
|
||||
assert_eq!(
|
||||
resolve_title_model("codex_cli", Some(" ")),
|
||||
Some(("gpt-6-luna".into(), TitleModelSource::Default))
|
||||
);
|
||||
unsafe { std::env::set_var(var, " env-model ") };
|
||||
assert_eq!(
|
||||
resolve_title_model("codex_cli", None),
|
||||
Some(("env-model".into(), TitleModelSource::Env))
|
||||
);
|
||||
assert_eq!(
|
||||
resolve_title_model("codex_cli", Some(" inst-model ")),
|
||||
Some(("inst-model".into(), TitleModelSource::Instance))
|
||||
);
|
||||
unsafe {
|
||||
match previous {
|
||||
Some(v) => std::env::set_var(var, v),
|
||||
None => std::env::remove_var(var),
|
||||
}
|
||||
}
|
||||
assert_eq!(resolve_title_model("gemini", Some("x")), None);
|
||||
assert_eq!(TitleModelSource::Env.as_str(), "env");
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load diff
|
|
@ -98,36 +98,6 @@ async fn main() -> Result<()> {
|
|||
),
|
||||
_ => {}
|
||||
}
|
||||
match archivr_core::database::fail_stalled_entry_summaries(&conn) {
|
||||
Ok(n) if n > 0 => eprintln!(
|
||||
"info: marked {n} stalled summary attempt(s) as failed in '{}'",
|
||||
archive.id
|
||||
),
|
||||
Err(e) => eprintln!(
|
||||
"warn: stalled summary cleanup failed for '{}': {e:#}",
|
||||
archive.id
|
||||
),
|
||||
_ => {}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Repair legacy bare-link titles on X Article tweets (idempotent; no-op once fixed).
|
||||
for archive in ®istry.archives {
|
||||
let Ok(paths) = archivr_core::archive::read_archive_paths(&archive.archive_path) else {
|
||||
continue;
|
||||
};
|
||||
match archivr_core::capture::backfill_x_article_titles(&paths) {
|
||||
Ok(0) => {}
|
||||
Ok(n) => eprintln!(
|
||||
"info: retitled {n} X Article entr{} in '{}'",
|
||||
if n == 1 { "y" } else { "ies" },
|
||||
archive.id
|
||||
),
|
||||
Err(e) => eprintln!(
|
||||
"warn: X Article title backfill failed for '{}': {e:#}",
|
||||
archive.id
|
||||
),
|
||||
}
|
||||
}
|
||||
|
||||
|
|
|
|||
File diff suppressed because it is too large
Load diff
47
crates/archivr-server/static/assets/index-B67momER.js
Normal file
47
crates/archivr-server/static/assets/index-B67momER.js
Normal file
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
1
crates/archivr-server/static/assets/index-DicH9MNh.css
Normal file
1
crates/archivr-server/static/assets/index-DicH9MNh.css
Normal file
File diff suppressed because one or more lines are too long
|
|
@ -6,8 +6,8 @@
|
|||
<title>Archivr</title>
|
||||
<link rel="icon" type="image/svg+xml" href="/favicon.svg">
|
||||
<link rel="icon" type="image/x-icon" href="/favicon.ico">
|
||||
<script type="module" crossorigin src="/assets/index-CKGxin5o.js"></script>
|
||||
<link rel="stylesheet" crossorigin href="/assets/index-CQKWq7vB.css">
|
||||
<script type="module" crossorigin src="/assets/index-B67momER.js"></script>
|
||||
<link rel="stylesheet" crossorigin href="/assets/index-DicH9MNh.css">
|
||||
</head>
|
||||
<body>
|
||||
<div id="root"></div>
|
||||
|
|
|
|||
|
|
@ -11,12 +11,6 @@ services:
|
|||
# Uncomment and set this to enable Twitter/X archiving.
|
||||
# The file must be accessible inside the container (e.g. in the config volume).
|
||||
# ARCHIVR_TWITTER_CREDENTIALS_FILE: /config/twitter-cookies.txt
|
||||
# Optional local transcription for YouTube summaries without subtitles.
|
||||
# Engines are not bundled: install them in a derived image or mount them.
|
||||
# ARCHIVR_TRANSCRIBE_ENGINES: "whisper,phonon2"
|
||||
# ARCHIVR_WHISPER_CLI: /models/bin/whisper-cli
|
||||
# ARCHIVR_WHISPER_MODEL: /models/ggml-large-v3-turbo.bin
|
||||
# ARCHIVR_PHONON2_CLI: /opt/transcribe/bin/fermion
|
||||
volumes:
|
||||
# Mount a directory containing archivr-server.toml as read-only config.
|
||||
# Copy docker/config.example.toml to ./config/archivr-server.toml to start.
|
||||
|
|
@ -24,8 +18,6 @@ services:
|
|||
# Persistent volume for the auth database and archive directories.
|
||||
# The paths inside must match archive_path values in your TOML config.
|
||||
- archivr-data:/data
|
||||
# Optional read-only model directory for local transcription engines.
|
||||
# - ./models:/models:ro
|
||||
|
||||
volumes:
|
||||
archivr-data:
|
||||
|
|
|
|||
428
docs/README.md
428
docs/README.md
|
|
@ -25,15 +25,9 @@ Archivr is a self-hosted tool for capturing and preserving digital content — Y
|
|||
- [Supported Inputs](#supported-inputs)
|
||||
- [YouTube playlists and channels](#youtube-playlists-and-channels)
|
||||
- [Video quality and audio-only](#video-quality-and-audio-only)
|
||||
- [YouTube subtitles](#youtube-subtitles)
|
||||
- [Text notes](#text-notes)
|
||||
- [Configuration](#configuration)
|
||||
- [TOML config file](#toml-config-file)
|
||||
- [Environment variables](#environment-variables)
|
||||
- [LLM providers](#llm-providers)
|
||||
- [Keeping yt-dlp and its JS runtime fresh](#keeping-yt-dlp-and-its-js-runtime-fresh)
|
||||
- [JavaScript runtime (Deno)](#javascript-runtime-deno)
|
||||
- [Troubleshooting](#troubleshooting)
|
||||
- [Deployment](#deployment)
|
||||
- [Security](#security)
|
||||
- [NixOS](#hosting-on-nixos)
|
||||
|
|
@ -43,17 +37,14 @@ Archivr is a self-hosted tool for capturing and preserving digital content — Y
|
|||
|
||||
## Features
|
||||
|
||||
- **Social media** — YouTube (videos, shorts, playlists, channels with sync mode; subtitles saved with each video by default), X/Twitter (tweet and thread JSON + media downloads), Instagram, TikTok, Facebook, Reddit, Snapchat via yt-dlp
|
||||
- **Social media** — YouTube (videos, shorts, playlists, channels with sync mode), X/Twitter (tweet and thread JSON + media downloads), Instagram, TikTok, Facebook, Reddit, Snapchat via yt-dlp
|
||||
- **Web pages** — full self-contained HTML snapshots via SingleFile + Chromium; optional Freedium mirror for paywalled articles; reader mode
|
||||
- **Local files** — import any file from disk by `file://` path
|
||||
- **Deduplication** — SHA3-256 content-addressed blob store shared across all captures; identical files are stored once
|
||||
- **Tags and search** — hierarchical tag tree, full-text search (including the latest completed summary and its generated JSON tags), filterable entry list
|
||||
- **Tags and search** — hierarchical tag tree, full-text search, filterable entry list
|
||||
- **Multiple archives** — the server mounts any number of separate archives from a single TOML config
|
||||
- **Role-based auth** — Guest / User / Admin / Owner roles; session cookies and API tokens; Argon2 passwords; the Owner can choose which roles (including custom ones) may reorder child entries
|
||||
- **Role-based auth** — Guest / User / Admin / Owner roles; session cookies and API tokens; Argon2 passwords
|
||||
- **Quality selection** — choose video quality or audio-only per capture; a live metadata probe populates the selector before download
|
||||
- **LLM summaries** — regenerable per-entry summary via the Anthropic HTTP API, an OpenAI-compatible HTTP API, a local `claude` CLI, or a local `codex` CLI; triggered manually from the entry rail, never automatically on capture; text-only by default, with an explicit `Include attached images` option; YouTube videos are summarized from their subtitles
|
||||
- **Text notes** — capture a plain-text or Markdown note with a title and no URL; the byte-preserving note is stored as a normal deduplicated blob and opens in the usual entry-rail preview
|
||||
- **In-progress capture indicator** — running captures appear as a compact spinner row in the entries list until they finish, replacing the earlier grey skeleton block
|
||||
|
||||
## Quick Start
|
||||
|
||||
|
|
@ -143,28 +134,12 @@ A separate auth database (`archivr-auth.sqlite`, path set in TOML) holds users,
|
|||
| Snapchat | Direct URL · `snapchat:ID` |
|
||||
| Arbitrary URL / web page | Any `https://` URL |
|
||||
|
||||
X Articles are titled `<article title> — @handle` from the article's own title instead of the tweet text (usually a
|
||||
bare `t.co` link). Article entries archived before this change are retitled on the next `archivr-server` start, but only
|
||||
while their title still equals the old auto-generated bare-link title; a renamed entry is left alone. Titles have no
|
||||
"edited" flag, so an entry you renamed back to exactly that old title is retitled too. CLI-only installs never run this
|
||||
pass — it runs at server startup only.
|
||||
|
||||
X threads keep `Thread by @handle` at capture. **Generate title** in the entry rail (thread entries, user role and up)
|
||||
asks the selected Summary provider's cheap title model for a short topic and saves `Thread about <topic> — @handle` as
|
||||
the entry title (`POST /api/archives/:id/entries/:uid/thread-title`, body `{"provider": "<kind>"}`); rename it like any
|
||||
title. With ≥2 entries selected and at least one X thread, the bulk panel's **Generate titles** does the same for each
|
||||
selected thread (non-threads skipped; uses the Summary provider selected in the rail), two at a time, with progress and
|
||||
an `X updated, Y failed` summary; changing the selection stops it picking up further entries. Models are listed under
|
||||
[LLM providers](#llm-providers).
|
||||
|
||||
### YouTube playlists and channels
|
||||
|
||||
Capturing a playlist or channel creates a **container entry** with each video archived as a child beneath it. Before downloading, the UI probes each video for available quality options — set quality per-video or apply one to the whole batch. Individual videos can be excluded with the remove button.
|
||||
|
||||
**Sync mode:** when re-archiving a playlist or channel, enable sync mode in the capture dialog to skip videos that are already in the archive. Only new videos are downloaded; the existing container is reused.
|
||||
|
||||
**Reordering:** if your role is allowed (by default Admin and Owner), expand a container on the main page and drag a video by its handle to change its position. On touch screens and phone-sized windows, where dragging isn't available, ↑/↓ buttons appear instead. Alt+↑/↓ on a selected row works everywhere. The order is saved to the archive. Videos added later by sync mode appear at the end. The Owner chooses which roles, including custom roles, may reorder under **Settings → Instance → Permissions**.
|
||||
|
||||
### Video quality and audio-only
|
||||
|
||||
When capturing a yt-dlp-backed source through the web UI, a metadata probe runs first and populates the quality selector with heights actually available in that video:
|
||||
|
|
@ -186,105 +161,6 @@ The `POST /api/archives/:id/captures` endpoint accepts an optional `quality` fie
|
|||
|
||||
`"audio"` selects the most efficient native audio track without re-encoding (Opus/WebM preferred, then AAC/M4A). Omitting `quality` or passing `"best"` downloads at the highest available quality.
|
||||
|
||||
### YouTube subtitles
|
||||
|
||||
YouTube video captures save subtitles next to the video by default: single videos, shorts, and every video archived
|
||||
from a YouTube playlist or channel, at any quality including audio-only. Subtitles apply to YouTube videos only —
|
||||
YouTube Music, Spotify, X, TikTok, and the other yt-dlp sources are downloaded without them.
|
||||
|
||||
At most two tracks are saved, chosen from the metadata yt-dlp already fetches for the capture:
|
||||
|
||||
- English, plus the video's original language when that isn't English.
|
||||
- Manual (uploader-provided) tracks are preferred. If the original language has no manual track, its auto-generated
|
||||
track is used; auto-generated English is used only when nothing else was found.
|
||||
- If the metadata probe fails, yt-dlp is asked for `en` and any `-orig` (original-language) track instead.
|
||||
|
||||
Tracks are requested as VTT, with SRT accepted; nothing is converted, and other subtitle formats are dropped. Each file
|
||||
goes through the usual SHA3-256 dedup into `store/raw/` and is recorded as a `subtitle` artifact of the entry, along
|
||||
with its language, whether it was manual or auto-generated, and its format.
|
||||
|
||||
Subtitle failures never fail a capture. Subtitles are requested in the same yt-dlp call as the media with
|
||||
`--ignore-errors`, so a missing track or a rate-limited caption request only logs a warning. If that call still fails,
|
||||
the media is retried once without subtitles.
|
||||
|
||||
Subtitles are on unless you turn them off for a capture:
|
||||
|
||||
- **Web UI:** the **Download subtitles** toggle in the capture dialog.
|
||||
- **API:** `"download_subtitles": false` in the `POST /api/archives/:id/captures` body. Omitting the field means `true`.
|
||||
- **CLI:** `archivr archive --no-subtitles <url>`.
|
||||
|
||||
Videos captured without subtitles can still be summarized; see [LLM providers](#llm-providers) and
|
||||
[Local transcription](#local-transcription-optional).
|
||||
|
||||
#### Local transcription (optional)
|
||||
|
||||
When a YouTube video has no subtitles at summary time, Archivr can transcribe its audio on the server. The order is
|
||||
fixed and transcription never runs if an earlier step yields usable subtitles:
|
||||
|
||||
1. archived subtitles;
|
||||
2. subtitles fetched from the original video (subtitles only, no media);
|
||||
3. local transcription with the engine chosen in the Summary panel;
|
||||
4. the no-subtitles error (or a transcription-specific error if step 3 ran and failed).
|
||||
|
||||
The feature is off until `ARCHIVR_TRANSCRIBE_ENGINES` lists at least one configured engine (env vars in
|
||||
[Local transcription env](#local-transcription)). Then the Summary panel shows a second selector on YouTube videos —
|
||||
**No local transcription** (default) or an enabled engine — remembered for the browser session. The API field is
|
||||
`"transcribe_engine": "<kind>"` in the summary POST body; an unknown or unconfigured engine is a 400. Enabled engines
|
||||
are listed by `GET /api/summary/transcription-engines`.
|
||||
|
||||
| Engine (`kind`) | Languages | Runs on | Install |
|
||||
|---|---|---|---|
|
||||
| Whisper (`whisper`) | ~99 (`*.en` models English only) | CPU, Metal, CUDA/Vulkan | whisper.cpp `whisper-cli` + a ggml model (default backend), or a faster-whisper wrapper (`script` backend) |
|
||||
| NVIDIA Parakeet (`parakeet`) | v2 English; v3 25 European | NVIDIA GPU (NeMo), Apple silicon (parakeet-mlx), CPU (ONNX) | your own wrapper script; weights CC-BY-4.0 |
|
||||
| Fermion Phonon-2 (`phonon2`) | **English only** (hard-coded) | Apple silicon (MLX), x86-64/Arm CPU, CUDA | `pip install fermion-research` + platform runtime; weights CC-BY-4.0, CLI licence unknown |
|
||||
|
||||
Setup examples:
|
||||
|
||||
```sh
|
||||
# whisper.cpp
|
||||
nix shell nixpkgs#whisper-cpp
|
||||
curl -LO https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-large-v3-turbo.bin
|
||||
export ARCHIVR_TRANSCRIBE_ENGINES=whisper ARCHIVR_WHISPER_MODEL=$PWD/ggml-large-v3-turbo.bin
|
||||
|
||||
# Parakeet via parakeet-mlx (Apple silicon), wrapper from spec Appendix A.3
|
||||
export ARCHIVR_TRANSCRIBE_ENGINES=parakeet ARCHIVR_PARAKEET_CLI=/opt/transcribe/parakeet.py
|
||||
|
||||
# Phonon-2 (vendor package; Apple silicon shown)
|
||||
python3 -m venv /opt/transcribe && /opt/transcribe/bin/pip install fermion-research mlx mlx-audio mlx-lm soundfile scipy zstandard
|
||||
export ARCHIVR_TRANSCRIBE_ENGINES=phonon2 ARCHIVR_PHONON2_CLI=/opt/transcribe/bin/fermion
|
||||
```
|
||||
|
||||
**Script contract** (Whisper `script` backend and Parakeet). Archivr runs
|
||||
`<script> --input <job>/audio.wav --output <job>/transcript.vtt --model <model> [--language <xx>]`. The script exits 0
|
||||
only after writing WebVTT to `--output`, may write a detected language code to `<output>.lang`, treats `--language` as
|
||||
a hint, writes nothing outside the output directory except model caches, and must tolerate SIGKILL on timeout. The
|
||||
audio is already 16 kHz mono PCM WAV. Reference wrappers are in the spec's Appendix A.
|
||||
|
||||
Notes:
|
||||
|
||||
- One transcription job runs at a time per server; others wait inside their own timeout budget.
|
||||
- Audio comes from the archived media or a yt-dlp audio download, converted by ffmpeg to a temp WAV (~115 MB per hour
|
||||
of audio) that is deleted afterwards.
|
||||
- `ARCHIVR_TRANSCRIBE_TIMEOUT` (default 3600 s) bounds the whole job: audio, ffmpeg and the engine.
|
||||
- The transcript is stored as a `subtitle` artifact with metadata `kind: "transcribed"`, `origin: "transcription"`,
|
||||
`engine` and `model` (a model path is reduced to its file name). It ranks below manual subtitles and above
|
||||
auto-generated ones, and later summaries reuse it without transcribing again.
|
||||
- Python engines download weights on first use, so the server user needs a writable `HOME`/cache directory.
|
||||
|
||||
Design and deviations: [`superpowers/specs/2026-10-05-local-transcription-fallback.md`](superpowers/specs/2026-10-05-local-transcription-fallback.md).
|
||||
|
||||
### Text notes
|
||||
|
||||
Not every capture has a URL. **Add text** in the capture dialog takes a title and a body and turns them into a
|
||||
self-contained entry — useful for a scrap of prose, a quote, or a note attached to the surrounding archive. The
|
||||
body is stored verbatim; no network fetch happens.
|
||||
|
||||
Two body types are accepted: `text/markdown` (saved as `.md`) and `text/plain` (saved as `.txt`). Anything else is
|
||||
rejected. The body lands in `store/raw/…` under its SHA3-256 content hash, exactly like every other capture, so an
|
||||
identical note captured twice is stored once.
|
||||
|
||||
Text notes have no synthetic source URL: the original-URL field stays empty rather than inventing a `text:` locator.
|
||||
|
||||
## Configuration
|
||||
|
||||
### TOML config file
|
||||
|
|
@ -315,11 +191,7 @@ See `docker/config.example.toml` for a complete annotated example.
|
|||
|---|---|---|
|
||||
| `ARCHIVR_BIND` | `127.0.0.1:8080` | Bind address; overrides `bind` in TOML |
|
||||
| `ARCHIVR_STATIC_DIR` | `crates/archivr-server/static` | Pre-built frontend asset directory |
|
||||
| `ARCHIVR_YT_DLP` | `yt-dlp` | yt-dlp binary used for video and social downloads; the Nix wrappers point this at the pinned release |
|
||||
| `ARCHIVR_YT_DLP_FORCE` | — | Absolute path to a yt-dlp binary that MUST be used, bypassing the resolver. Prefer `ARCHIVR_YT_DLP` unless you are overriding for a specific run |
|
||||
| `ARCHIVR_DENO` | — | Pinned Deno binary offered to the JS runtime resolver; the Nix wrappers and Docker image set it |
|
||||
| `ARCHIVR_JS_RUNTIME` | — | `RUNTIME[:ABS_PATH]` with `RUNTIME` one of `deno`, `node`, `bun`, `quickjs`. Forces the runtime passed to yt-dlp, bypassing resolution and version checks. Invalid values are warned about and ignored. Node needs ≥ 22; Bun needs ≥ 1.2.11 and is deprecated in yt-dlp |
|
||||
| `ARCHIVR_STATE_DIR` | platform state dir | Where `archivr yt-dlp update` and Settings › Instance › yt-dlp install yt-dlp and Deno. Docker sets `/data/archivr-state` |
|
||||
| `ARCHIVR_YT_DLP` | `yt-dlp` | yt-dlp binary used for video and social downloads |
|
||||
| `ARCHIVR_SINGLE_FILE` | `single-file` | single-file-cli binary for web page archiving |
|
||||
| `ARCHIVR_CHROME` | `chromium` | Chromium executable passed to single-file |
|
||||
| `ARCHIVR_CHROME_ARGS` | — | Extra space-separated Chromium flags (Docker sets `--no-sandbox`) |
|
||||
|
|
@ -327,248 +199,7 @@ See `docker/config.example.toml` for a complete annotated example.
|
|||
| `ARCHIVR_TWEET_SCRAPER` | `vendor/twitter/scrape_user_tweet_contents.py` | Tweet scraper script path |
|
||||
| `ARCHIVR_TWEET_PYTHON` | `python3` | Python executable for the tweet scraper |
|
||||
|
||||
The Nix wrapper and Docker image set `ARCHIVR_STATIC_DIR`, `ARCHIVR_SINGLE_FILE`, `ARCHIVR_CHROME`, `ARCHIVR_DENO`, and
|
||||
`ARCHIVR_FFMPEG` automatically.
|
||||
|
||||
#### LLM providers
|
||||
|
||||
Summaries are manual and provider-agnostic. Only the variables for the provider you actually select are read; the two
|
||||
HTTP providers refuse to start without their API key. They are text-only by default. Selecting `Include attached images`
|
||||
explicitly sends eligible archived image data to the chosen provider; it is never attached automatically.
|
||||
|
||||
| Provider | Attached images |
|
||||
|---|---|
|
||||
| Anthropic HTTP | Supported |
|
||||
| OpenAI-compatible HTTP | Supported |
|
||||
| Codex CLI | Supported |
|
||||
| Claude CLI | Not supported |
|
||||
|
||||
Image inclusion considers only `media` artifacts with `jpg`, `jpeg`, `png`, `webp`, `gif`, or `avif` files. At most four
|
||||
images are sent, each no larger than 5 MiB and no more than 12 MiB in total.
|
||||
|
||||
**YouTube videos** are summarized from one archived subtitle track, never from the video file. The track is picked in
|
||||
this order: manual English, manual original language, other manual, auto-generated original language, auto-generated
|
||||
English. It is reduced to plain text: timestamps, cue settings, and markup are removed, and the repeated lines of
|
||||
rolling auto-captions are collapsed. Like every summary input, the transcript is cut at 48,000 characters, so the end of
|
||||
a long video is not summarized. Adding or replacing subtitles changes the input hash, so a cached summary built from
|
||||
different subtitles is not reused. Other video and audio entries still can't be summarized.
|
||||
|
||||
If the entry has no usable subtitles (for example, it was captured with subtitles turned off, or by an older Archivr),
|
||||
requesting a summary first downloads them from the original URL — subtitles only, no media — while the attempt shows as
|
||||
pending. Fetched tracks are archived to the entry like captured ones. If none can be fetched, the attempt fails without
|
||||
calling the provider and the Summary panel shows:
|
||||
|
||||
> This video can’t be summarized because no subtitles are available. Archivr found no archived subtitles and couldn’t
|
||||
> download any from the original video — it may have no captions, or it may be private, deleted, or unreachable.
|
||||
|
||||
If a transcription engine was selected, Archivr transcribes the audio before giving up; see
|
||||
[Local transcription](#local-transcription-optional).
|
||||
|
||||
Free-text entry search also matches the latest completed summary text and its generated JSON tags. Entries with no
|
||||
summary, or only a pending or failed summary, get no summary-derived match.
|
||||
|
||||
Each request is cached under the provider and the **requested** model identifier. If a provider reports a more precise
|
||||
resolved model (for example, an alias's concrete version), the UI displays that resolved name as attribution without
|
||||
changing the cache identity.
|
||||
|
||||
Summary attempts move from `pending` to `running` and then to `completed` or `failed`. On server startup, interrupted
|
||||
pending or running attempts are marked failed. Regenerating does not replace an earlier completed summary until the
|
||||
replacement succeeds, and public readers receive completed content only—never pending state or diagnostic errors.
|
||||
|
||||
| Variable | Default | Description |
|
||||
|---|---|---|
|
||||
| `ARCHIVR_ANTHROPIC_API_KEY` | *(required for `anthropic_http`)* | API key for the Anthropic Messages API |
|
||||
| `ARCHIVR_ANTHROPIC_URL` | `https://api.anthropic.com/v1/messages` | Endpoint override, e.g. an internal proxy |
|
||||
| `ARCHIVR_ANTHROPIC_MODEL` | `claude-3-5-sonnet-latest` | Model id used for Anthropic summaries |
|
||||
| `ARCHIVR_OPENAI_API_KEY` | *(required for `openai_compatible`)* | API key for any OpenAI-compatible endpoint |
|
||||
| `ARCHIVR_OPENAI_URL` | `https://api.openai.com/v1/chat/completions` | Endpoint override; point this at a local server to run offline |
|
||||
| `ARCHIVR_OPENAI_MODEL` | `gpt-4o-mini` | Model id used for OpenAI-compatible summaries |
|
||||
| `ARCHIVR_CLAUDE_CLI` | *(auto-discovered)* | Path to a local `claude` binary |
|
||||
| `ARCHIVR_CLAUDE_MODEL` | *(the CLI's own default)* | Optional model override for the local Claude CLI |
|
||||
| `ARCHIVR_CODEX_CLI` | *(auto-discovered)* | Path to a local `codex` binary |
|
||||
| `ARCHIVR_CODEX_MODEL` | *(the CLI's own default)* | Optional model override for the local Codex CLI |
|
||||
| `ARCHIVR_SUMMARY_HTTP_TIMEOUT` | `120` | Seconds before an HTTP-provider summary is killed |
|
||||
| `ARCHIVR_SUMMARY_CLI_TIMEOUT` | `300` | Seconds before a CLI-provider summary is killed; also bounds each summary-time yt-dlp subtitle fetch call |
|
||||
|
||||
Thread-title generation uses the same provider but never its summary model (`ARCHIVR_*_MODEL`); it uses a cheap title
|
||||
model instead. Admins can set a per-provider title model in **Settings › Instance › Thread title models**. Precedence:
|
||||
that instance setting (blank = unset) > the env var below > the built-in default.
|
||||
|
||||
| Variable | Default | Description |
|
||||
|---|---|---|
|
||||
| `ARCHIVR_ANTHROPIC_TITLE_MODEL` | `claude-haiku-4-5` | Model used only for thread-title generation |
|
||||
| `ARCHIVR_OPENAI_TITLE_MODEL` | `gpt-4o-mini` | Model used only for thread-title generation |
|
||||
| `ARCHIVR_CLAUDE_TITLE_MODEL` | `haiku` | Model used only for thread-title generation |
|
||||
| `ARCHIVR_CODEX_TITLE_MODEL` | `gpt-6-luna` | Model used only for thread-title generation; set this if your Codex account lacks that model |
|
||||
|
||||
When `ARCHIVR_CLAUDE_CLI` / `ARCHIVR_CODEX_CLI` is unset the binary is auto-discovered, in this order: the well-known
|
||||
absolute paths, then `$HOME/.local/bin/<name>`, then the bare name resolved through `PATH`. Note that `PATH` is
|
||||
consulted **last** — if a stale binary sits at one of the well-known paths it wins over a newer one on `PATH`, so set
|
||||
the variable explicitly when you have both. The well-known paths are `/opt/homebrew/bin/claude` and
|
||||
`/usr/local/bin/claude` for Claude, and `/Applications/ChatGPT.app/Contents/Resources/codex`,
|
||||
`/opt/homebrew/bin/codex`, and `/usr/local/bin/codex` for Codex.
|
||||
|
||||
#### Local transcription
|
||||
|
||||
Only read when a summary request names an engine. See [Local transcription](#local-transcription-optional).
|
||||
|
||||
| Variable | Default | Description |
|
||||
|---|---|---|
|
||||
| `ARCHIVR_TRANSCRIBE_ENGINES` | *(unset: feature off)* | Comma-separated enabled engines: `whisper`, `parakeet`, `phonon2`. Unknown names are warned about and ignored |
|
||||
| `ARCHIVR_WHISPER_BACKEND` | `whisper_cpp` | `whisper_cpp` or `script` |
|
||||
| `ARCHIVR_WHISPER_CLI` | auto-discovered `whisper-cli` | whisper.cpp binary; **required** with the `script` backend (the wrapper script) |
|
||||
| `ARCHIVR_WHISPER_MODEL` | *(required for `whisper`)* | whisper.cpp: ggml model file; script: passed as `--model` |
|
||||
| `ARCHIVR_WHISPER_LANGUAGES` | *(any)* | Optional allowlist of base language codes, e.g. `en` for a `*.en` model |
|
||||
| `ARCHIVR_PARAKEET_CLI` | *(required for `parakeet`)* | Wrapper script following the script contract |
|
||||
| `ARCHIVR_PARAKEET_MODEL` | `nvidia/parakeet-tdt-0.6b-v3` | Passed as `--model` |
|
||||
| `ARCHIVR_PARAKEET_LANGUAGES` | *(any)* | Optional allowlist; use `en` for v2 |
|
||||
| `ARCHIVR_PHONON2_CLI` | auto-discovered `fermion` | The `fermion` CLI |
|
||||
| `ARCHIVR_PHONON2_MODEL` | `phonon-2` | Model passed to `fermion transcribe` |
|
||||
| `ARCHIVR_TRANSCRIBE_TIMEOUT` | `3600` | Seconds for one whole transcription job |
|
||||
| `ARCHIVR_FFMPEG` | `ffmpeg` | ffmpeg used to extract 16 kHz mono WAV; Nix wrappers and Docker set it |
|
||||
|
||||
`whisper-cli` and `fermion` are auto-discovered like the Claude/Codex CLIs: `/opt/homebrew/bin/<name>`,
|
||||
`/usr/local/bin/<name>`, `$HOME/.local/bin/<name>`, then `PATH`.
|
||||
|
||||
## Keeping yt-dlp and its JS runtime fresh
|
||||
|
||||
yt-dlp is the download engine behind every video and social capture. YouTube rotates its player-signature and API
|
||||
surfaces on a days-to-weeks cadence, so a binary that worked last month starts returning HTTP 403 on downloads. Keeping
|
||||
it current is ordinary maintenance, not an emergency.
|
||||
|
||||
**What ships.** `flake.nix` pins a specific yt-dlp release fetched straight from `github.com/yt-dlp/yt-dlp/releases`,
|
||||
not from nixpkgs — that channel usually lags months behind. The `archivr-server` and `archivr` wrappers set
|
||||
`ARCHIVR_YT_DLP` to that pinned binary.
|
||||
|
||||
**How the resolver picks.** At runtime archivr probes `--version` on each candidate — the pinned binary from
|
||||
`ARCHIVR_YT_DLP` and any user-installed binary at `<state_dir>/yt-dlp/yt-dlp` — and runs the newest. An exact version
|
||||
tie resolves in favour of your own install. Setting `ARCHIVR_YT_DLP_FORCE=/path/to/yt-dlp` bypasses the comparison
|
||||
entirely. The state dir is `~/Library/Application Support/archivr` on macOS, and `$XDG_STATE_HOME/archivr` (default
|
||||
`~/.local/state/archivr`) elsewhere; `ARCHIVR_STATE_DIR` overrides it.
|
||||
|
||||
There are three ways to get a fresh version, cheapest first. From the web UI, **Settings › Instance › yt-dlp** (admins
|
||||
only) shows every yt-dlp and JS runtime candidate with its version and the winner, and **Update yt-dlp & Deno** runs
|
||||
the same update as the CLI. The server picks up the new binaries immediately — no restart; captures already running keep the binary
|
||||
they started with. Admin API: `GET /api/admin/yt-dlp` (status) and `POST /api/admin/yt-dlp/update` (no body; per
|
||||
component outcome plus fresh status; 409 while an update is running).
|
||||
|
||||
**1. Self-update — no rebuild required.**
|
||||
|
||||
```sh
|
||||
archivr yt-dlp status # every candidate, its version, and which one wins
|
||||
archivr yt-dlp update # download the latest zipapp and the latest Deno into the state dir
|
||||
archivr yt-dlp update --version 2026.09.15 # pin a specific release tag
|
||||
```
|
||||
|
||||
When `ARCHIVR_YT_DLP_FORCE` applies, `status` shows that forced candidate and selects it as the winner.
|
||||
|
||||
`update` installs yt-dlp and Deno independently and reports each (`yt-dlp: …`, `deno: …`); it exits non-zero if
|
||||
either failed. `--version` applies to yt-dlp only — Deno always tracks the latest release. The CLI runs in its own
|
||||
process, so restart a running server after a CLI update; an update from the web UI needs no restart.
|
||||
|
||||
The released artifact is a Python zipapp, so this path needs Python ≥ 3.10 on `PATH` as `python3` at run time. The web
|
||||
UI update runs the installed yt-dlp once and reports it as failed, with the error, if it does not start (e.g. macOS's
|
||||
system Python 3.9); `status` shows a candidate that exists but does not run as `invalid: <last error line>`.
|
||||
|
||||
**2. Automatic weekly bump.** `.github/workflows/update-ytdlp.yml` runs every Monday at 06:00 UTC, queries GitHub for
|
||||
the latest release, and opens a PR bumping `version` and `hash` in `flake.nix` via `peter-evans/create-pull-request`.
|
||||
It also accepts `workflow_dispatch` for an on-demand run.
|
||||
|
||||
**3. Manual bump**, when you need it now and do not want to wait for the weekly:
|
||||
|
||||
```sh
|
||||
NEW=$(curl -s https://api.github.com/repos/yt-dlp/yt-dlp/releases/latest | jq -r .tag_name)
|
||||
HASH=$(nix hash file --sri --type sha256 <(curl -sL "https://github.com/yt-dlp/yt-dlp/releases/download/${NEW}/yt-dlp"))
|
||||
|
||||
# In flake.nix, inside the `ytDlp = pkgs.stdenv.mkDerivation { … }` block:
|
||||
# version = "OLD"; → version = "$NEW";
|
||||
# url = ".../download/OLD/yt-dlp"; → .../download/$NEW/yt-dlp
|
||||
# hash = "sha256-OLD…"; → hash = "$HASH";
|
||||
|
||||
nix build .#archivr-server
|
||||
./result/bin/archivr yt-dlp status # the env row should report the new version
|
||||
git commit -am "chore(nix): yt-dlp OLD → $NEW"
|
||||
```
|
||||
|
||||
**4. Docker:** yt-dlp is pinned to a specific version in the `Dockerfile` (`pip install "yt-dlp[default]==<version>"`, which also pulls the `yt-dlp-ejs` challenge solver), matching the Nix pin. To update the baked-in copy, bump the version string in the `Dockerfile` venv install step to match the new Nix version, then rebuild:
|
||||
|
||||
```sh
|
||||
docker build -t archivr-server .
|
||||
docker compose up -d
|
||||
```
|
||||
|
||||
Self-update also works in the container: the image sets `ARCHIVR_STATE_DIR=/data/archivr-state` on the persistent
|
||||
`archivr-data` volume, so the update survives restarts. Use Settings › Instance › yt-dlp (no restart), or the CLI and
|
||||
then restart so the server re-resolves:
|
||||
|
||||
```sh
|
||||
docker compose exec archivr archivr yt-dlp update
|
||||
docker compose restart archivr
|
||||
```
|
||||
|
||||
Deno is pinned in the `Dockerfile` (2.9.7, sha256 per arch). Nothing bumps it automatically: change the version and
|
||||
both sha256 values together. It adds roughly 80 MB to the image.
|
||||
|
||||
### JavaScript runtime (Deno)
|
||||
|
||||
YouTube now serves player challenges that yt-dlp solves with its EJS solver, which needs a JavaScript runtime —
|
||||
Deno ≥ 2.3.0 by default. Without one, downloads fail with HTTP 403.
|
||||
|
||||
**How the resolver picks.** `ARCHIVR_JS_RUNTIME` wins outright when valid. Otherwise archivr probes `deno --version`
|
||||
on the pinned `ARCHIVR_DENO` and on `<state_dir>/deno/deno`, skips anything below 2.3.0, and uses the newest (exact
|
||||
ties go to the state-dir copy). If neither qualifies it falls back to `deno` on `PATH`; if that fails too, it prints a
|
||||
one-time `warn: no JavaScript runtime for yt-dlp …` and runs yt-dlp without one. Only Deno is chosen automatically;
|
||||
Node, Bun and QuickJS are used only when forced. Every yt-dlp call gets the result as `--js-runtimes deno:<path>`
|
||||
(non-Deno runtimes also get `--no-js-runtimes` first so a stray Deno cannot outrank them).
|
||||
|
||||
`archivr yt-dlp status` prints a second table after the yt-dlp one. Columns are tab-separated; locations are absolute
|
||||
(the state dir is `$XDG_STATE_HOME/archivr`, default `~/.local/state/archivr`, on Linux and
|
||||
`~/Library/Application Support/archivr` on macOS). Exactly one row — the role the resolver picked — is starred, even
|
||||
when two roles point at the same binary. On Linux after `archivr yt-dlp update`, with an older Deno on `PATH`:
|
||||
|
||||
```text
|
||||
JS runtime (passed to yt-dlp as --js-runtimes)
|
||||
role path version chosen
|
||||
force (ARCHIVR_JS_RUNTIME) — —
|
||||
env (ARCHIVR_DENO) — —
|
||||
state-dir /home/alice/.local/state/archivr/deno/deno 2.9.7 *
|
||||
path (deno) /usr/bin/deno 2.4.0
|
||||
```
|
||||
|
||||
Under the Nix wrappers the pinned Deno is also on `PATH`, so two rows show the same binary but only the pin is chosen:
|
||||
|
||||
```text
|
||||
JS runtime (passed to yt-dlp as --js-runtimes)
|
||||
role path version chosen
|
||||
force (ARCHIVR_JS_RUNTIME) — —
|
||||
env (ARCHIVR_DENO) /nix/store/…-deno-2.9.4/bin/deno 2.9.4 *
|
||||
state-dir — —
|
||||
path (deno) /nix/store/…-deno-2.9.4/bin/deno 2.9.4
|
||||
```
|
||||
|
||||
An invalid `ARCHIVR_JS_RUNTIME` shows as `invalid: <reason>` in the force row.
|
||||
|
||||
### Troubleshooting
|
||||
|
||||
**`WARNING: [youtube] No supported JavaScript runtime could be found…` followed by `HTTP Error 403: Forbidden`.**
|
||||
yt-dlp ran without a JS runtime.
|
||||
|
||||
1. Run `archivr yt-dlp status` (or open Settings › Instance › yt-dlp) and check the JS runtime table has a chosen row.
|
||||
2. If not, run `archivr yt-dlp update` or **Update yt-dlp & Deno** in the UI (installs Deno into the state dir), or set
|
||||
`ARCHIVR_JS_RUNTIME=node:/abs/node` (Node ≥ 22).
|
||||
3. After a CLI update, restart `archivr-server`; a UI update re-resolves automatically. An env-var change always
|
||||
needs a restart.
|
||||
4. Verify by hand: `yt-dlp -v --js-runtimes deno:<path> --simulate <url>` should print `JS runtimes: deno-…`.
|
||||
|
||||
**`no such option: --js-runtimes`.** A yt-dlp older than 2025.11 (only reachable through the bare `PATH` fallback)
|
||||
does not know the flag. Run `archivr yt-dlp update`.
|
||||
|
||||
**Deno runs but the solver still fails.** yt-dlp treats any Deno stderr output as a solver error. Deno writes its
|
||||
cache to `DENO_DIR` (default `$HOME/.cache/deno`); with a read-only `HOME`, point `DENO_DIR` at a writable directory.
|
||||
|
||||
**NixOS: `update` says the prebuilt Deno cannot execute.** Upstream Deno binaries are dynamically linked and need a
|
||||
standard loader. Enable `programs.nix-ld`, or rely on the pinned `ARCHIVR_DENO` from the flake — `update` then skips
|
||||
Deno and exits 0 (the UI shows the Deno outcome as `skipped: …`).
|
||||
The Nix wrapper and Docker image set `ARCHIVR_STATIC_DIR`, `ARCHIVR_SINGLE_FILE`, and `ARCHIVR_CHROME` automatically.
|
||||
|
||||
## Deployment
|
||||
|
||||
|
|
@ -612,29 +243,6 @@ Set `openFirewall = true` with a non-loopback `listenAddress` only when LAN or r
|
|||
|
||||
Archive directories must be owned by the `archivr` user. Initialise them with `archivr init` first, then `chown -R archivr:archivr /srv/archivr`.
|
||||
|
||||
The wrappers ship nixpkgs' Deno as `ARCHIVR_DENO`, so YouTube works out of the box. `archivr yt-dlp update` downloads
|
||||
the upstream (dynamically linked) Deno, which only runs on NixOS with `programs.nix-ld.enable = true`; without it,
|
||||
`update` skips Deno, keeps the pinned one, and exits 0. Run `update` as the service user so it lands in
|
||||
`/var/lib/archivr-server`, then restart the unit — or use Settings › Instance › yt-dlp, which runs as the service user
|
||||
and needs no restart.
|
||||
|
||||
Extra environment (LLM providers, local transcription) goes through two options:
|
||||
|
||||
```nix
|
||||
services.archivr-server = {
|
||||
environment = {
|
||||
ARCHIVR_TRANSCRIBE_ENGINES = "whisper";
|
||||
ARCHIVR_WHISPER_CLI = "${pkgs.whisper-cpp}/bin/whisper-cli";
|
||||
ARCHIVR_WHISPER_MODEL = "/var/lib/archivr-server/models/ggml-large-v3-turbo.bin";
|
||||
};
|
||||
environmentFile = "/run/secrets/archivr.env"; # KEY=value lines, e.g. API keys; kept out of the Nix store
|
||||
};
|
||||
```
|
||||
|
||||
`environment` is merged over defaults (`lib.mkDefault`) that put `HOME`, `XDG_CACHE_HOME` and `HF_HOME` under
|
||||
`/var/lib/archivr-server`, so Python engines can cache weights. The unit sets no `PrivateDevices`/`DeviceAllow`;
|
||||
adding them would break CUDA engines.
|
||||
|
||||
### Hosting with Docker
|
||||
|
||||
```sh
|
||||
|
|
@ -666,21 +274,6 @@ environment:
|
|||
ARCHIVR_TWITTER_CREDENTIALS_FILE: /config/twitter-cookies.txt
|
||||
```
|
||||
|
||||
**Local transcription:** the image ships no engines (it sets `ARCHIVR_FFMPEG=/usr/bin/ffmpeg`). `docker-compose.yml`
|
||||
has commented `ARCHIVR_TRANSCRIBE_ENGINES`/`ARCHIVR_WHISPER_*`/`ARCHIVR_PHONON2_CLI` examples and a read-only
|
||||
`./models:/models:ro` mount. For Phonon-2 on CPU, build a derived image:
|
||||
|
||||
```dockerfile
|
||||
FROM archivr:latest
|
||||
RUN python3 -m venv /opt/transcribe && \
|
||||
/opt/transcribe/bin/pip install --no-deps torch --index-url https://download.pytorch.org/whl/cpu && \
|
||||
/opt/transcribe/bin/pip install fermion-research torch safetensors soundfile scipy zstandard
|
||||
ENV ARCHIVR_TRANSCRIBE_ENGINES=phonon2 ARCHIVR_PHONON2_CLI=/opt/transcribe/bin/fermion
|
||||
```
|
||||
|
||||
Mount a cache volume at the server user's `~/.cache` so downloaded weights survive container recreation. GPU
|
||||
containers need the NVIDIA container toolkit and a CUDA base image.
|
||||
|
||||
**Building locally:**
|
||||
|
||||
```sh
|
||||
|
|
@ -691,12 +284,7 @@ The image compiles the Rust binary in a separate build stage; only runtime depen
|
|||
|
||||
## Development
|
||||
|
||||
Runtime dependencies beyond Rust and Node: `yt-dlp`, Deno (≥ 2.3.0, for YouTube), Chromium, `single-file` (Node), Python 3 with `twitter-api-client`, `ffmpeg`. `nix develop` provides the dev subset.
|
||||
|
||||
Entry summaries are served by one of four interchangeable providers — `anthropic_http`, `openai_compatible`,
|
||||
`claude_cli`, or `codex_cli` — each configured entirely through the environment; see
|
||||
[LLM providers](#llm-providers) for the full variable list. The `archivr` CLI itself exposes `archive`, `init`, and
|
||||
`yt-dlp status` / `yt-dlp update`; summaries are triggered from the web UI rather than the command line.
|
||||
Runtime dependencies beyond Rust and Node: `yt-dlp`, Chromium, `single-file` (Node), Python 3 with `twitter-api-client`, `ffmpeg`. `nix develop` provides the dev subset.
|
||||
|
||||
```sh
|
||||
# Rust (workspace root)
|
||||
|
|
@ -708,7 +296,8 @@ cargo run -p archivr-server -- ./archivr-server.toml
|
|||
# Frontend (from frontend/)
|
||||
bun install
|
||||
bun run dev # Vite dev server
|
||||
bun run build # → crates/archivr-server/static/ (gitignored; nix build does this automatically)
|
||||
bun run build # → crates/archivr-server/static/
|
||||
bun run storybook # Component QA on :6006
|
||||
|
||||
# Nix
|
||||
nix develop # dev shell
|
||||
|
|
@ -718,4 +307,3 @@ nix build .#archivr-server
|
|||
## License
|
||||
|
||||
MIT — see [LICENSE](../LICENSE.md).
|
||||
\n
|
||||
|
|
|
|||
|
|
@ -1,474 +0,0 @@
|
|||
# X Article, Vision Summaries, and Summary Search Implementation Plan
|
||||
|
||||
> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
|
||||
|
||||
**Goal:** Make X Articles summarize their archived article text, let a user explicitly attach eligible archived images to a requested summary, and search the latest completed summary text (including its JSON tags).
|
||||
|
||||
**Architecture:** Keep `archivr-core` synchronous and make the summary input the single carrier of both reduced text and an opt-in, bounded list of image descriptors. The image-selection policy is serialized into the existing `input_sha256` preimage, preserving the existing `entry_summaries` uniqueness key without a migration. Extend the existing server-side entry search query with a correlated latest-completed-summary predicate; the endpoint response and frontend search transport stay unchanged.
|
||||
|
||||
**Tech Stack:** Rust 2024, `anyhow`, `rusqlite`, `reqwest` blocking HTTP, `serde_json`, Axum, React JSX, and plain CSS.
|
||||
|
||||
---
|
||||
|
||||
## File map and interfaces
|
||||
|
||||
| File | Responsibility |
|
||||
| --- | --- |
|
||||
| `crates/archivr-core/src/summarizer.rs` | X Article reducer, image candidate policy, `SummaryBuildOptions`, `SummaryImage`, cache digest preimage, provider request payloads, Codex invocation, and unit tests. |
|
||||
| `crates/archivr-server/src/routes.rs` | Parse `include_images`, reject an unsupported Claude CLI vision request before a job is created, and pass build options to both cache lookup and the background worker. |
|
||||
| `frontend/src/api.js` | Send the explicit `include_images` boolean in the existing summary POST. |
|
||||
| `frontend/src/components/ContextRail.jsx` | Per-generation checkbox, provider-specific disabled Claude state, privacy/cap warning, and request wiring. |
|
||||
| `frontend/src/styles.css` | Dedicated summary-image option layout and disabled-note treatment. |
|
||||
| `crates/archivr-core/src/archive.rs` | Correlated SQL predicate for the latest completed summary and search tests. |
|
||||
| `docs/README.md`, `AGENTS.md`, `ARCHIVR-MENTAL-MODEL.md` | User, contributor, and architectural documentation after the implementation is complete. |
|
||||
|
||||
Define the following core interfaces before server or UI tasks use them. `archive_file` is an absolute local path derived from `ArchivePaths.store_path` plus the stored artifact `relpath`; it is never returned from an API.
|
||||
|
||||
```rust
|
||||
pub const MAX_SUMMARY_IMAGES: usize = 4;
|
||||
pub const MAX_SUMMARY_IMAGE_BYTES: u64 = 5 * 1024 * 1024;
|
||||
pub const MAX_SUMMARY_IMAGE_TOTAL_BYTES: u64 = 12 * 1024 * 1024;
|
||||
|
||||
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
|
||||
pub struct SummaryBuildOptions {
|
||||
pub include_images: bool,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct SummaryImage {
|
||||
pub sha256: String,
|
||||
pub mime_type: String,
|
||||
pub byte_size: u64,
|
||||
pub archive_file: PathBuf,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct SummaryRequest {
|
||||
pub entry_uid: String,
|
||||
pub title: Option<String>,
|
||||
pub source_kind: String,
|
||||
pub entity_kind: String,
|
||||
pub content: String,
|
||||
pub images: Vec<SummaryImage>,
|
||||
}
|
||||
|
||||
pub fn build_summary_input(
|
||||
paths: &ArchivePaths,
|
||||
entry_uid: &str,
|
||||
options: SummaryBuildOptions,
|
||||
) -> Result<SummaryInput>;
|
||||
|
||||
pub fn summarize_entry(
|
||||
archive_paths: &ArchivePaths,
|
||||
entry_uid: &str,
|
||||
options: SummaryBuildOptions,
|
||||
provider: &dyn SummaryProvider,
|
||||
prompt_version: &str,
|
||||
) -> Result<database::EntrySummaryRecord>;
|
||||
```
|
||||
|
||||
Use one deterministic digest preimage: `content` bytes, then `"\0images="`, then `include_images` as `"0"` or `"1"`, followed by each selected image in query order as `"\0" + sha256 + "\0" + mime_type + "\0" + byte_size`. Hash that complete byte sequence with the existing `hash::hash_bytes`. This means an unchecked request has no images and a different digest from a checked request even when no candidate qualifies.
|
||||
|
||||
### Task 1: Add the X Article reducer with test-first precedence
|
||||
|
||||
**Files:**
|
||||
|
||||
- Modify: `crates/archivr-core/src/summarizer.rs`
|
||||
- Test: `crates/archivr-core/src/summarizer.rs` (`#[cfg(test)] mod tests`)
|
||||
|
||||
- [ ] **Step 1: Write failing tests for each per-status X Article precedence rule.**
|
||||
|
||||
Add assertions against `extract_tweet_text` using values that retain a link-only top-level tweet body:
|
||||
|
||||
```rust
|
||||
#[test]
|
||||
fn extract_tweet_text_prefers_x_article_plain_text_over_tco_body() {
|
||||
let tweet = serde_json::json!({
|
||||
"full_text": "https://t.co/article",
|
||||
"article": { "title": "Skin guide", "plain_text": "Use sunscreen daily." }
|
||||
});
|
||||
assert_eq!(extract_tweet_text(&tweet).as_deref(),
|
||||
Some("Skin guide\n\nUse sunscreen daily."));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn extract_tweet_text_uses_article_blocks_when_plain_text_is_empty() {
|
||||
let tweet = serde_json::json!({"article": {
|
||||
"title": "Blocks", "plain_text": " ",
|
||||
"blocks": [{"text": "First"}, {"children": [{"text": "Second"}]}]
|
||||
}});
|
||||
assert_eq!(extract_tweet_text(&tweet).as_deref(), Some("Blocks\n\nFirst\n\nSecond"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn extract_tweet_text_falls_back_from_article_to_link_only_tweet_body() {
|
||||
let tweet = serde_json::json!({
|
||||
"full_text": "https://t.co/fallback",
|
||||
"article": {"title": "Preview", "preview_text": "Preview copy", "summary_text": "Later"}
|
||||
});
|
||||
assert_eq!(extract_tweet_text(&tweet).as_deref(), Some("Preview\n\nPreview copy"));
|
||||
}
|
||||
```
|
||||
|
||||
- [ ] **Step 2: Run the focused test target and observe it fail.**
|
||||
|
||||
Run: `cargo test -p archivr-core extract_tweet_text_`
|
||||
|
||||
Expected: FAIL because the current reducer returns the top-level `full_text` or does not descend into `article.blocks`.
|
||||
|
||||
- [ ] **Step 3: Implement `article_text` and deterministic block flattening.**
|
||||
|
||||
Add private helpers before `extract_tweet_text`:
|
||||
|
||||
```rust
|
||||
fn nonempty_string(v: &serde_json::Value, key: &str) -> Option<String>;
|
||||
fn flatten_article_blocks(v: &serde_json::Value, out: &mut Vec<String>);
|
||||
fn article_text(status: &serde_json::Value) -> Option<String>;
|
||||
```
|
||||
|
||||
`article_text` must inspect the status's `article` object before ordinary fields. With a nonempty title, format each successful source as `title + "\n\n" + body`; if title is empty, return only `body`. Select the body in this exact order: nonblank `plain_text`; recursive text leaves from `blocks` in JSON array/object encounter order; nonblank `preview_text`; nonblank `summary_text`. `flatten_article_blocks` must collect only textual scalar values from conventional textual keys (`text`, `plain_text`, `content`, `body`, `title`, `heading`) and recursively visit arrays and objects; it must not stringify IDs, URLs, booleans, media metadata, or arbitrary scalar fields. Join block leaves with `"\n\n"`.
|
||||
|
||||
- [ ] **Step 4: Integrate the helper into all tweet-status paths.**
|
||||
|
||||
Make `one(v)` call `article_text(v).or_else(|| normal_tweet_text(v))`, where `normal_tweet_text` preserves the existing `full_text`, `text`, `content`, `body` sequence. Keep support for a top-level `{ "tweet": ... }` wrapper and for embedded `thread`, `tweets`, and `replies` members.
|
||||
|
||||
- [ ] **Step 5: Add the thread-artifact regression test and run the focused tests.**
|
||||
|
||||
Add a fixture archive with two `raw_tweet_json` artifacts, where each JSON status has article text, then assert `build_summary_input(..., SummaryBuildOptions::default())?.request.content` contains both article bodies separated by `"\n\n---\n\n"`. Run: `cargo test -p archivr-core extract_tweet_text_ build_summary_input_`
|
||||
|
||||
Expected: PASS, including the existing wrapped/thread tweet tests.
|
||||
|
||||
- [ ] **Step 6: Commit the atomic reducer change.**
|
||||
|
||||
```bash
|
||||
git add crates/archivr-core/src/summarizer.rs
|
||||
git commit -m "fix: summarize X Article text"
|
||||
```
|
||||
|
||||
### Task 2: Model and select explicit image inputs in core
|
||||
|
||||
**Files:**
|
||||
|
||||
- Modify: `crates/archivr-core/src/summarizer.rs`
|
||||
- Test: `crates/archivr-core/src/summarizer.rs` (`#[cfg(test)] mod tests`)
|
||||
|
||||
- [ ] **Step 1: Write failing candidate-selection and digest tests.**
|
||||
|
||||
Build a temporary archive entry containing `media` artifacts for valid `jpg`, `png`, `webp`, `gif`, and `avif`, plus `avatar`, `video`, `audio`, unsupported `svg`, one 5 MiB + 1 byte image, and enough valid images to exceed both the four-image and 12 MiB limits. Assert only role `media`, allowed MIME/extension pairs, at most four descriptors, no descriptor over 5 MiB, and total selected bytes at most 12 MiB. Also assert:
|
||||
|
||||
```rust
|
||||
let text_only = build_summary_input(&paths, &uid, SummaryBuildOptions { include_images: false })?;
|
||||
let visual = build_summary_input(&paths, &uid, SummaryBuildOptions { include_images: true })?;
|
||||
assert!(text_only.request.images.is_empty());
|
||||
assert!(!visual.request.images.is_empty());
|
||||
assert_ne!(text_only.input_sha256, visual.input_sha256);
|
||||
```
|
||||
|
||||
- [ ] **Step 2: Run the new core tests and observe failure.**
|
||||
|
||||
Run: `cargo test -p archivr-core summary_image_`
|
||||
|
||||
Expected: FAIL because `SummaryRequest` has no images and input construction has no image-selection mode.
|
||||
|
||||
- [ ] **Step 3: Define the shared image model and query candidates from existing blobs.**
|
||||
|
||||
Add the constants and `SummaryBuildOptions`, `SummaryImage`, and `SummaryRequest.images` definitions from the file map. Add a private `load_summary_image_candidates(conn, entry_id) -> Result<Vec<SummaryImage>>` querying `entry_artifacts ea JOIN blobs b` for `ea.entry_id = ?1 AND ea.artifact_role = 'media'`, ordered by `ea.id ASC`, selecting `b.sha256`, `b.mime_type`, `b.extension`, `b.byte_size`, and `ea.relpath`.
|
||||
|
||||
Accept a candidate only when both of the following are true: its extension is one of `jpg`, `jpeg`, `png`, `webp`, `gif`, `avif`, and its MIME is the matching `image/jpeg`, `image/png`, `image/webp`, `image/gif`, or `image/avif` family. Resolve `archive_file` under the configured store path and reject a candidate whose canonicalized/normalized path escapes that store root. Stop at the first candidate that would exceed either image count, per-image, or aggregate byte limit; continue scanning later candidates so a too-large or unsupported early artifact cannot hide a valid later one.
|
||||
|
||||
- [ ] **Step 4: Build options-aware input and a complete cache digest.**
|
||||
|
||||
Change every current `build_summary_input` call to pass `SummaryBuildOptions::default()` until Task 4 changes the server. Populate `request.images` only if `options.include_images` is true; text extraction and `MAX_INPUT_CHARS` handling remain identical. Replace `hash_bytes(content.as_bytes())` with a private `summary_input_digest(content, include_images, images)` implementing the stated NUL-delimited preimage so the existing database uniqueness constraint continues to distinguish all modes. Do not alter `database.rs`: `input_sha256` already participates in the cache key.
|
||||
|
||||
- [ ] **Step 5: Run core regression tests.**
|
||||
|
||||
Run: `cargo test -p archivr-core summary_image_ build_summary_input_`
|
||||
|
||||
Expected: PASS; the text-only request has zero image descriptors, and selection is deterministic by artifact insertion order.
|
||||
|
||||
- [ ] **Step 6: Commit the core input model.**
|
||||
|
||||
```bash
|
||||
git add crates/archivr-core/src/summarizer.rs
|
||||
git commit -m "feat: model opt-in summary images"
|
||||
```
|
||||
|
||||
### Task 3: Make provider transports honor image descriptors
|
||||
|
||||
**Files:**
|
||||
|
||||
- Modify: `crates/archivr-core/src/summarizer.rs`
|
||||
- Test: `crates/archivr-core/src/summarizer.rs` (`#[cfg(test)] mod tests`)
|
||||
|
||||
- [ ] **Step 1: Write failing payload, command, and capability tests.**
|
||||
|
||||
Construct a `SummaryRequest` with one tiny fixture `SummaryImage` and assert `anthropic_request_body` puts a text block and `{ "type": "image", "source": { "type": "base64", "media_type": "image/png", "data": "..." } }` in `messages[0].content`. Assert `openai_request_body` emits a text content part plus `{ "type": "image_url", "image_url": { "url": "data:image/png;base64,..." } }`. Unit-test a pure Codex argument builder so its primary arguments contain `exec`, `--image`, the fixture path, `--output-last-message`, output path, and final `-`; test its positional fallback also keeps `--image`. Assert `ClaudeCliProvider::summarize` returns an error containing `Claude CLI cannot attach local images` when `request.images` is nonempty.
|
||||
|
||||
- [ ] **Step 2: Run the provider tests and observe failure.**
|
||||
|
||||
Run: `cargo test -p archivr-core "anthropic_request_body|openai_request_body|codex.*image|claude.*images"`
|
||||
|
||||
Expected: FAIL because HTTP bodies are string-only, Codex has no `--image`, and Claude silently accepts the request.
|
||||
|
||||
- [ ] **Step 3: Encode images for each HTTP protocol.**
|
||||
|
||||
Add `read_image_base64(image: &SummaryImage) -> Result<String>` which reads only the already bounded selected file and uses `base64::Engine` with the existing dependency or workspace dependency. Make `anthropic_request_body` and `openai_request_body` return their present text-only JSON shapes when `request.images.is_empty()` and their documented content-part arrays otherwise. Preserve `SYSTEM_PROMPT`, `build_user_prompt`, provider URL, headers, and response parsing.
|
||||
|
||||
- [ ] **Step 4: Extend Codex safely and reject Claude at the provider boundary.**
|
||||
|
||||
Refactor `codex::run` to accept `&[SummaryImage]`; add `--image <archive_file>` once per selected image before `--output-last-message` in both primary stdin and positional-prompt forms. Keep the existing last-message temporary-file contract and cleanup behavior. Have `ClaudeCliProvider::summarize` `bail!("Claude CLI cannot attach local images; choose an HTTP provider or Codex CLI")` before spawning when images are supplied.
|
||||
|
||||
- [ ] **Step 5: Thread image-aware options through synchronous orchestration.**
|
||||
|
||||
Change `summarize_entry` to accept `SummaryBuildOptions` and call the options-aware builder before provider invocation. This is a synchronous core function; do not introduce Tokio, async traits, or new database fields.
|
||||
|
||||
- [ ] **Step 6: Run provider and existing summary tests.**
|
||||
|
||||
Run: `cargo test -p archivr-core summarizer::tests`
|
||||
|
||||
Expected: PASS, with all four providers retaining their text-only behavior when `images` is empty.
|
||||
|
||||
- [ ] **Step 7: Commit the provider implementation.**
|
||||
|
||||
```bash
|
||||
git add crates/archivr-core/src/summarizer.rs
|
||||
git commit -m "feat: attach opted-in images to summaries"
|
||||
```
|
||||
|
||||
### Task 4: Expose the opt-in flag through the server API
|
||||
|
||||
**Files:**
|
||||
|
||||
- Modify: `crates/archivr-server/src/routes.rs`
|
||||
- Test: `crates/archivr-server/src/routes.rs` (`#[cfg(test)] mod tests`)
|
||||
|
||||
- [ ] **Step 1: Write failing route tests.**
|
||||
|
||||
Add authenticated POST tests that deserialize a request without `include_images` and assert it takes the text-only build path, and with `{ "provider": "claude_cli", "include_images": true }` assert status `400 BAD_REQUEST` and an error containing `Claude CLI cannot attach local images`. Add a successful non-Claude request test with `include_images: true` using a configured test provider and assert the returned cache key differs from the equivalent text-only request. Keep GET assertions unchanged: it remains `{ "entry_uid", "summary" }`.
|
||||
|
||||
- [ ] **Step 2: Run the focused route tests and observe failure.**
|
||||
|
||||
Run: `cargo test -p archivr-server "summary.*include_images|claude.*images"`
|
||||
|
||||
Expected: FAIL because the request body does not accept the field and no capability check occurs.
|
||||
|
||||
- [ ] **Step 3: Add the backward-compatible request field and capability guard.**
|
||||
|
||||
Extend `SummaryRequestBody` exactly as follows:
|
||||
|
||||
```rust
|
||||
#[derive(Debug, serde::Deserialize)]
|
||||
struct SummaryRequestBody {
|
||||
provider: String,
|
||||
#[serde(default)]
|
||||
force: bool,
|
||||
#[serde(default)]
|
||||
include_images: bool,
|
||||
}
|
||||
```
|
||||
|
||||
After resolving `provider_cfg` and before cache lookup/upsert/spawn, return `ApiError::bad_request("Claude CLI cannot attach local images; choose an HTTP provider or Codex CLI")` when `body.include_images && matches!(provider_cfg, ProviderConfig::ClaudeCli(_))`. Construct `SummaryBuildOptions { include_images: body.include_images }` once and pass it to the preflight builder and cloned into `spawn_blocking` for `summarize_entry`.
|
||||
|
||||
- [ ] **Step 4: Verify cache and lifecycle consistency.**
|
||||
|
||||
Ensure preflight `build_summary_input` and background `summarize_entry` receive the same options, so `find_entry_summary` and `upsert_pending_entry_summary` use the same digest. Do not change `entry_summaries`, `latest_entry_summary`, or the GET route; the existing input-hash uniqueness constraint is sufficient.
|
||||
|
||||
- [ ] **Step 5: Run the focused server tests.**
|
||||
|
||||
Run: `cargo test -p archivr-server "summary.*include_images|claude.*images"`
|
||||
|
||||
Expected: PASS; omitting the new field is text-only and a Claude image request produces no pending summary row.
|
||||
|
||||
- [ ] **Step 6: Commit the API wiring.**
|
||||
|
||||
```bash
|
||||
git add crates/archivr-server/src/routes.rs
|
||||
git commit -m "feat: accept image summary requests"
|
||||
```
|
||||
|
||||
### Task 5: Add the explicit, provider-aware image consent control
|
||||
|
||||
**Files:**
|
||||
|
||||
- Modify: `frontend/src/api.js`
|
||||
- Modify: `frontend/src/components/ContextRail.jsx`
|
||||
- Modify: `frontend/src/styles.css`
|
||||
- Test: manual browser smoke test (no frontend test harness exists)
|
||||
|
||||
- [ ] **Step 1: Inspect the existing provider selector and write the manual failure script.**
|
||||
|
||||
In a locally authenticated entry detail, select an HTTP provider and verify the Summary section currently has no `Include attached images` checkbox; select Claude CLI and verify there is no capability explanation. Record this as the observed pre-implementation failure. Do not add a frontend test framework.
|
||||
|
||||
- [ ] **Step 2: Change the API client contract.**
|
||||
|
||||
Change the function signature and POST body only:
|
||||
|
||||
```js
|
||||
export async function requestEntrySummary(
|
||||
archiveId, entryUid, { provider, force = false, includeImages = false } = {}
|
||||
) {
|
||||
// existing fetch and error parsing
|
||||
body: JSON.stringify({ provider, force, include_images: includeImages })
|
||||
}
|
||||
```
|
||||
|
||||
Keep all `fetch` calls inside `frontend/src/api.js`; do not add an inline fetch in the component.
|
||||
|
||||
- [ ] **Step 3: Implement the local per-generation control.**
|
||||
|
||||
Add `const [includeSummaryImages, setIncludeSummaryImages] = useState(false)` beside the summary provider state. In `handleGenerateSummary`, pass `includeImages: includeSummaryImages`. Reset this state to `false` whenever `detail?.summary?.entry_uid` changes, so a consent choice cannot carry to another entry. Do not persist the checkbox in `sessionStorage`; the consent is per generation and defaults off.
|
||||
|
||||
- [ ] **Step 4: Render clear consent, scope, and Claude capability states.**
|
||||
|
||||
In `.rail-summary-controls`, below the provider `<select>`, render a labeled checkbox with exact visible label `Include attached images`. Its help text must say that selected archived images are sent to the chosen provider and that only up to four supported images (5 MiB each, 12 MiB total) can be attached; unsupported or oversized artifacts are skipped. When `summaryProvider === 'claude_cli'`, render the checkbox disabled, force `includeSummaryImages` to false via an effect or provider-change handler, and show `Claude CLI cannot attach local images. Choose an HTTP provider or Codex CLI.` Do not submit a silently dropped image choice.
|
||||
|
||||
- [ ] **Step 5: Add scoped plain-CSS rules.**
|
||||
|
||||
Add `.rail-summary-image-option`, `.rail-summary-image-option__label`, `.rail-summary-image-option__note`, and `.rail-summary-image-option--disabled` under the existing summary rail CSS. Use the project variables (`--muted`, `--line`, `--paper`) and preserve keyboard focus and normal checkbox semantics; do not use a generic row class or inline layout styles.
|
||||
|
||||
- [ ] **Step 6: Run the manual success script and build verification.**
|
||||
|
||||
Run: `bun run build`
|
||||
|
||||
Expected: successful production bundle in `crates/archivr-server/static`. Then manually verify: unchecked generation sends `include_images:false`; checked Anthropic/OpenAI/Codex generation sends `true`; changing to Claude unchecks/disables the control and shows the exact explanation; server errors still appear through existing `summaryError` handling.
|
||||
|
||||
- [ ] **Step 7: Commit source files, not generated static output.**
|
||||
|
||||
```bash
|
||||
git add frontend/src/api.js frontend/src/components/ContextRail.jsx frontend/src/styles.css
|
||||
git commit -m "feat: add summary image consent control"
|
||||
```
|
||||
|
||||
### Task 6: Search latest completed summary JSON without changing API shape
|
||||
|
||||
**Files:**
|
||||
|
||||
- Modify: `crates/archivr-core/src/archive.rs`
|
||||
- Test: `crates/archivr-core/src/archive.rs` (`#[cfg(test)] mod tests`)
|
||||
|
||||
- [ ] **Step 1: Write failing search tests with real cache rows.**
|
||||
|
||||
Extend `make_test_db_with_entries` or add a focused fixture helper that inserts summary rows through `database::upsert_pending_entry_summary` and `database::update_entry_summary_status`. Add assertions for all of the following:
|
||||
|
||||
```rust
|
||||
// A completed JSON string with {"tags":["skincare","dermatology"]} matches skincare.
|
||||
// An unrelated query yields no result.
|
||||
// Two completed rows: the newer completed row is searched; the older one is not.
|
||||
// A completed older row remains matched while a newer row is pending or failed.
|
||||
// source:, entity:, url:, title:, after:, before:, tag: and collection/visibility scope keep their current behavior.
|
||||
```
|
||||
|
||||
Set distinct `updated_at` values (or insert/transition rows in distinct timestamp order) so "latest completed" is unambiguous. The tag assertion must match `summary_text` itself, not `entry_tag_assignments`.
|
||||
|
||||
- [ ] **Step 2: Run the focused tests and observe failure.**
|
||||
|
||||
Run: `cargo test -p archivr-core search_.*summary`
|
||||
|
||||
Expected: FAIL because free text only checks entry and source identity fields.
|
||||
|
||||
- [ ] **Step 3: Add a parameter-bound latest-completed summary predicate.**
|
||||
|
||||
In the existing unqualified `query.q` block in `search_entries`, preserve every present `LOWER(...) LIKE ?{n}` condition and add this clause using the same single bound `term`:
|
||||
|
||||
```sql
|
||||
OR LOWER(COALESCE((
|
||||
SELECT s.summary_text
|
||||
FROM entry_summaries s
|
||||
WHERE s.entry_id = e.id
|
||||
AND s.status = 'completed'
|
||||
AND s.summary_text IS NOT NULL
|
||||
ORDER BY s.completed_at DESC, s.updated_at DESC, s.id DESC
|
||||
LIMIT 1
|
||||
), '')) LIKE ?N
|
||||
```
|
||||
|
||||
Use the existing numbered parameter construction (`?{n}`) and push `term` once, so user text is never concatenated into SQL. `completed_at` ordering means only completed rows participate, and a later pending/failed row cannot displace an older completed row. Do not add an archive method, schema column, migration, route parameter, or frontend response field.
|
||||
|
||||
- [ ] **Step 4: Run core search tests.**
|
||||
|
||||
Run: `cargo test -p archivr-core search_`
|
||||
|
||||
Expected: PASS, including old prefix-filter behavior and JSON tag substring matches.
|
||||
|
||||
- [ ] **Step 5: Verify the existing API transport needs no change.**
|
||||
|
||||
Inspect `crates/archivr-server/src/routes.rs::search_entries_handler`, `frontend/src/api.js::searchEntries`, and `frontend/src/App.jsx` search call sites. Confirm the server still calls `archive::search_entries` with the same `SearchEntriesQuery` and returns `Vec<EntrySummary>`; record no source edit for these files unless the inspection reveals a type break. Add no client-side filtering.
|
||||
|
||||
- [ ] **Step 6: Commit the search change.**
|
||||
|
||||
```bash
|
||||
git add crates/archivr-core/src/archive.rs
|
||||
git commit -m "feat: search completed summary tags"
|
||||
```
|
||||
|
||||
### Task 7: Document, bundle, and verify the completed feature set
|
||||
|
||||
**Files:**
|
||||
|
||||
- Modify: `docs/README.md`
|
||||
- Modify: `AGENTS.md`
|
||||
- Modify: `ARCHIVR-MENTAL-MODEL.md`
|
||||
- Generated (do not hand-edit): `crates/archivr-server/static/`
|
||||
|
||||
- [ ] **Step 1: Update user documentation.**
|
||||
|
||||
In `docs/README.md`, document that summaries are manual, text-only by default, and the explicit `Include attached images` option sends at most four eligible local images to the selected provider. State the allowed formats and byte limits, identify Anthropic/OpenAI-compatible/Codex support, state Claude CLI cannot attach local images, and state free-text search includes the latest completed summary text and its generated tags.
|
||||
|
||||
- [ ] **Step 2: Update contributor constraints.**
|
||||
|
||||
In `AGENTS.md`, record the `SummaryBuildOptions`/digest rule, the image candidate role and limits, the provider capability matrix, and the latest-completed-only search semantic. Preserve the rule that core remains synchronous and that generated static files are not hand-edited.
|
||||
|
||||
- [ ] **Step 3: Update the architectural data-flow documentation.**
|
||||
|
||||
In `ARCHIVR-MENTAL-MODEL.md`, extend the LLM Summary section to show explicit UI consent flowing into image selection, cache hashing, provider transport, and the existing row lifecycle. Add that entry search reads only the latest completed `summary_text`, retaining a prior completed result while newer work is pending or failed.
|
||||
|
||||
- [ ] **Step 4: Build and test the final implementation.**
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
cargo test
|
||||
bun --cwd frontend run build
|
||||
cargo build
|
||||
```
|
||||
|
||||
Expected: all Rust tests pass, frontend build succeeds, and the generated static bundle contains the checkbox UI. Do not hand-edit generated files; include them in a commit only if this repository currently tracks frontend bundle changes after `bun run build`.
|
||||
|
||||
- [ ] **Step 5: Run the end-to-end manual smoke test.**
|
||||
|
||||
Start the server with a test archive and verify: an X Article whose normal tweet text is only a t.co URL summarizes article body text; a text-only generation remains unchanged; a checked vision-capable request attaches only bounded eligible images; Claude has a disabled explanatory option and a direct API request is rejected; a `skincare` search finds completed summary JSON tags; pending/failed rows do not hide a previous completed match.
|
||||
|
||||
- [ ] **Step 6: Commit documentation and any tracked generated bundle.**
|
||||
|
||||
```bash
|
||||
git add docs/README.md AGENTS.md ARCHIVR-MENTAL-MODEL.md crates/archivr-server/static
|
||||
git commit -m "docs: explain image summaries and summary search"
|
||||
```
|
||||
|
||||
### Task 8: Final implementation review before integration
|
||||
|
||||
**Files:**
|
||||
|
||||
- Review: `crates/archivr-core/src/summarizer.rs`
|
||||
- Review: `crates/archivr-core/src/archive.rs`
|
||||
- Review: `crates/archivr-server/src/routes.rs`
|
||||
- Review: `frontend/src/api.js`
|
||||
- Review: `frontend/src/components/ContextRail.jsx`
|
||||
- Review: `frontend/src/styles.css`
|
||||
- Review: `docs/README.md`, `AGENTS.md`, `ARCHIVR-MENTAL-MODEL.md`
|
||||
|
||||
- [ ] **Step 1: Perform the approved-design coverage review.**
|
||||
|
||||
Verify A is covered by Tasks 1 and 2 (plain text, blocks, preview/summary, normal tweet fallback, and every thread JSON artifact); B by Tasks 2–5 (explicit default-off consent, selection policy/caps, input hash, all four provider outcomes, POST flag, and compatible GET); and C by Task 6 (latest completed summary JSON/tags, pending/failed semantics, prefix-filter preservation, server-side architecture).
|
||||
|
||||
- [ ] **Step 2: Scan the plan and implementation for unfinished markers and type drift.**
|
||||
|
||||
Run: `rg -n -i '\\bt[o]do\\b|\\bt[b]d\\b|placehold[e]r|implement[[:space:]]later' docs/superpowers/plans/2026-08-23-x-article-vision-search.md crates/archivr-core/src/summarizer.rs crates/archivr-core/src/archive.rs crates/archivr-server/src/routes.rs frontend/src`
|
||||
|
||||
Expected: no newly introduced unfinished markers in the changed feature code or plan. Confirm every use of `SummaryBuildOptions`, `SummaryImage`, options-aware `build_summary_input`, and options-aware `summarize_entry` matches the Task 2 definitions.
|
||||
|
||||
- [ ] **Step 3: Review commits and working tree.**
|
||||
|
||||
Run: `git log --oneline --decorate -8` and `git status --short`
|
||||
|
||||
Expected: atomic commits cover the reducer, core image model/providers, server API, frontend control, search, and docs; no unintended artifacts or source edits remain.
|
||||
|
|
@ -1,874 +0,0 @@
|
|||
# Spec: local transcription fallback for YouTube summaries
|
||||
|
||||
- **Status:** Implemented (2026-10-05). See "Implementation deviations" at the end for where the code differs from this text.
|
||||
- **Date:** 2026-10-05
|
||||
- **Depends on:** the YouTube subtitle capture and subtitle-based summarization feature (subtitle artifacts, `crates/archivr-core/src/subtitles.rs`, fetching subtitles on demand at summary time, and the `NoSubtitlesAvailable` failure). The symbols named below come from that feature.
|
||||
- **Audience:** a model or engineer who implements this without any other context. Read `AGENTS.md` and `ARCHIVR-MENTAL-MODEL.md` first. The repo rules apply throughout:
|
||||
- core stays synchronous
|
||||
- errors are `anyhow`
|
||||
- logging uses `eprintln!` with `info:`/`warn:` prefixes
|
||||
- external tools are configured by `ARCHIVR_*` env vars, never TOML
|
||||
- yt-dlp processes are only built with `yt_dlp_command()` (resolver-chosen binary plus `--js-runtimes`)
|
||||
- frontend API calls only go through `frontend/src/api.js`
|
||||
- CSS is plain
|
||||
- tests are in-file `#[cfg(test)]` modules
|
||||
|
||||
Markers used in this document:
|
||||
- **[INFERENCE]**: a claim about third-party software that was *not* run while writing this spec. The implementer must check it against the installed version before relying on it.
|
||||
- **[UNKNOWN]**: information the sources did not provide.
|
||||
|
||||
---
|
||||
|
||||
## 1. Goal and non-goals
|
||||
|
||||
### Goal
|
||||
A summary is requested for a `youtube`/`video` entry. No subtitle artifact is archived, and fetching subtitles on demand from the original video adds none. Today that ends in `NO_SUBTITLES_SUMMARY_MESSAGE`. With this feature, Archivr can **optionally transcribe the audio locally** with an engine the user picks:
|
||||
|
||||
| Engine kind | Label | Languages |
|
||||
|---|---|---|
|
||||
| `whisper` | Whisper (whisper.cpp natively, or faster-whisper or any other Whisper runtime through a wrapper script) | multilingual |
|
||||
| `parakeet` | NVIDIA Parakeet (`parakeet-tdt-0.6b-v2`/`-v3` through a wrapper script around NeMo, parakeet-mlx, sherpa-onnx, …) | English (v2) or 25 European languages (v3) |
|
||||
| `phonon2` | Fermion Research Phonon-2 | **English only** |
|
||||
|
||||
The transcript is stored as a normal `subtitle` artifact with `kind: "transcribed"`, so the existing ranking, reduction and digest code turns it into summary input without any special cases. Later summaries of the same entry reuse it and do not transcribe again.
|
||||
|
||||
### Non-goals
|
||||
- **No cloud ASR.** Everything runs as a local subprocess on the server host. HTTP transcription APIs (including Phonon's own `fermion serve` OpenAI-compatible endpoint) are out of scope; see §12.
|
||||
- **No automatic transcription at capture time.** It only happens when a user asks for a summary and picks an engine in that request. A capture-time option is a possible later extension (§12).
|
||||
- **No transcription for non-YouTube media.** Other video and audio entries still get `UNSUPPORTED_SUMMARY_CONTENT_MESSAGE`. Extending to them is §12.
|
||||
- **Archivr does not bundle models or Python engine runtimes.** Models are large and licensed separately. Users install engines and point env vars at them. The deployment changes (§10) only wire up `ffmpeg` and pass env vars through.
|
||||
- **No re-transcription UI or engine switching** for an entry that already has a transcript (§12).
|
||||
- **No word-level timestamps, diarization or translation.** The output is a plain cue-level VTT.
|
||||
- **No new TOML config.**
|
||||
|
||||
---
|
||||
|
||||
## 2. Background: what exists after the subtitle feature
|
||||
|
||||
These symbols are the shared contracts of the subtitle feature. Reuse them; do not reimplement them.
|
||||
|
||||
| Area | Symbol | Behaviour relevant here |
|
||||
|---|---|---|
|
||||
| `downloader/ytdlp.rs` | `SubtitleKind { Manual, Auto, Unknown }`, `as_str`, `parse` | The kind is persisted in artifact `metadata_json.kind`. |
|
||||
| | `StagedSubtitle { path, language, kind, format, original_language }` | A staged sidecar file in `store/temp/<key>/`. |
|
||||
| | `plan_subtitle_request(metadata_json) -> Option<SubtitleRequest>` | Derives `original_language` from `--dump-json` (`language` field, or else an auto key ending `-orig`). |
|
||||
| | `pub(crate) language_base(code)` | Lowercases, strips `-orig`, keeps the first `-` segment (`"de-orig"` → `"de"`, `"en-GB"` → `"en"`). |
|
||||
| | private `is_safe_language_code(code)` | `^[A-Za-z0-9][A-Za-z0-9-]*$`. |
|
||||
| | `fetch_metadata(url, cookies) -> Option<String>`, `fetch_metadata_with_timeout(url, cookies, timeout)` | `--dump-json`; `None` when the video can't be reached or the bound expires. Summary-time calls (and `download_subtitles`) are bounded by `ARCHIVR_SUMMARY_CLI_TIMEOUT`. |
|
||||
| | `resolve_yt_dlp()`, `yt_dlp_command(&ytdlp)` | `resolve_yt_dlp()` picks the binary; `yt_dlp_command()` is the only way to build a yt-dlp `Command` (adds the resolved `--js-runtimes` args). |
|
||||
| `downloader/store.rs` | `archive_staged_file(file, store_path) -> Result<PathBuf>` | SHA3 content-addressed move into `raw/`. |
|
||||
| `subtitles.rs` | `SUBTITLE_ARTIFACT_ROLE = "subtitle"`, `SUBTITLE_ORIGIN_CAPTURE`, `SUBTITLE_ORIGIN_SUMMARY_FETCH` | Role and origin strings. |
|
||||
| | `SubtitleFormat { Vtt, Srt }` with `detect`, `mime`, `extension` | |
|
||||
| | `ArchivedSubtitle { raw_relpath, language, kind, format, original_language }` | |
|
||||
| | `archive_staged_subtitles(store_path, staged) -> Vec<ArchivedSubtitle>` | Per-file errors are logged and skipped. |
|
||||
| | `register_subtitle_artifacts(conn, store_path, entry_id, subs, origin) -> Result<usize>` | IMMEDIATE transaction; skips an existing `(entry, "subtitle", blob)` and logs/skips a file it can't stat. |
|
||||
| | `fetch_subtitles_for_entry(paths, entry_uid, cookie_rules) -> Result<usize>` | Fetches on demand. Returns early with `Ok(0)` for non-YouTube or non-`http(s)` entries, and with `Ok(n)` when a usable (non-empty) subtitle track already exists. |
|
||||
| | `subtitle_to_transcript`, `parse_subtitle_metadata`, `subtitle_track_rank` | Reducer, metadata parse, ranking. |
|
||||
| `database.rs` | `entry_source_info`, `list_entry_artifacts_by_role`, `entry_has_artifact_blob`, `update_entry_summary_input_sha256` | |
|
||||
| `summarizer.rs` | `NO_SUBTITLES_SUMMARY_MESSAGE`, `SUBTITLE_FETCH_PENDING_INPUT_SHA256`, `is_no_subtitles_error`, `build_summary_input_with_subtitle_fetch(paths, entry_uid, options, cookie_rules)` | Background fetch-then-build entry point. |
|
||||
| | private `run_cli(executable, args, prompt, timeout_secs)` | Thread + channel watchdog: the child is killed when `recv_timeout` expires. |
|
||||
| | `required_env`, `env_or`, `optional_env`, `env_timeout`, `resolve_cli` (private) | Env-resolution helpers. |
|
||||
| `routes.rs` | `request_entry_summary_handler` with `PreflightOutcome::{Cached, Pending, FetchSubtitles}`; `summary_failure_error_text`; `record_background_summary_failure` | Server flow. |
|
||||
| `ContextRail.jsx` | `SUMMARY_PROVIDERS`, `SUMMARY_PROVIDER_KEY` sessionStorage, the `.rail-summary-controls` block, the "Generating…" spinner, 1500 ms polling, the failed-attempt `<p className="form-msg form-msg--err rail-summary-error">` | UI. |
|
||||
|
||||
The ranking table from the subtitle feature (D4), which this spec extends in §7:
|
||||
|
||||
| Rank | Track |
|
||||
|---|---|
|
||||
| 0 | manual + en |
|
||||
| 1 | manual + orig |
|
||||
| 2 | other manual |
|
||||
| 3 | auto/unknown + orig |
|
||||
| 4 | auto/unknown + en |
|
||||
| 5 | anything else |
|
||||
|
||||
---
|
||||
|
||||
## 3. Where it plugs in
|
||||
|
||||
Transcription runs **inside the background summary worker**, between fetching subtitles on demand and the final `NoSubtitlesAvailable` error. During that time the summary row stays `pending`, holding the placeholder `input_sha256 = SUBTITLE_FETCH_PENDING_INPUT_SHA256`. That is the same row lifecycle the subtitle fetch already uses, so the UI shows the existing "Generating…" spinner and polls every 1500 ms.
|
||||
|
||||
Transcription runs only when **all** of these hold:
|
||||
1. The POST body names an engine in `transcribe_engine`. That engine is listed in `ARCHIVR_TRANSCRIBE_ENGINES` and fully configured. Both are checked synchronously at preflight, before any row is created; a failure is a 400.
|
||||
2. The entry is `youtube`/`video`. `build_summary_input` returned `NoSubtitlesAvailable` at preflight, so the handler took the `PreflightOutcome::FetchSubtitles` branch.
|
||||
3. After `fetch_subtitles_for_entry`, `build_summary_input` **still** returns `NoSubtitlesAvailable`. Rebuilding is the ground truth: the fetch can add nothing, or add only tracks that reduce to empty text, and both cases must lead to transcription.
|
||||
4. The engine accepts the video's language (§6.5).
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
POST["POST /summary {provider, transcribe_engine?}"] --> CFG{"provider_from_env + transcriber::request_from_env (if engine given)"}
|
||||
CFG -- "error" --> E400["400 naming the env var"]
|
||||
CFG -- ok --> PRE["build_summary_input (preflight)"]
|
||||
PRE -- "Ok(input)" --> SYNC["existing path: cache lookup / pending row / provider"]
|
||||
PRE -- "NoSubtitlesAvailable" --> ROW["pending row, input_sha256 = pending-subtitle-fetch, 202"]
|
||||
ROW --> BG["spawn_blocking: build_summary_input_with_subtitle_fetch(.., transcription)"]
|
||||
BG --> FETCH["subtitles::fetch_subtitles_for_entry -> SubtitleFetchOutcome"]
|
||||
FETCH --> RB1{"build_summary_input"}
|
||||
RB1 -- "Ok" --> SUM["update_entry_summary_input_sha256 + summarize_prebuilt_entry"]
|
||||
RB1 -- "NoSubtitlesAvailable, no engine" --> FAIL1["failed row: NO_SUBTITLES_SUMMARY_MESSAGE"]
|
||||
RB1 -- "NoSubtitlesAvailable, engine requested" --> GATE{"engine supports original language?"}
|
||||
GATE -- no --> FAIL2["failed row: language-unsupported copy"]
|
||||
GATE -- yes --> TR["transcriber::transcribe_entry: audio -> ffmpeg 16 kHz mono WAV -> engine -> VTT -> register subtitle artifact (kind transcribed)"]
|
||||
TR -- "error / timeout" --> FAIL3["failed row: sanitized transcription copy"]
|
||||
TR -- ok --> RB2{"build_summary_input"}
|
||||
RB2 -- "Ok" --> SUM
|
||||
RB2 -- "NoSubtitlesAvailable" --> FAIL4["failed row: NO_SUBTITLES_AFTER_TRANSCRIPTION_MESSAGE"]
|
||||
```
|
||||
|
||||
No provider/LLM call happens on any failure path.
|
||||
|
||||
### 3.1 Changed entry point
|
||||
|
||||
Change the signature of `summarizer::build_summary_input_with_subtitle_fetch` and migrate its single caller in `routes.rs`. This is a clean cutover: no second function and no shim.
|
||||
|
||||
```rust
|
||||
pub fn build_summary_input_with_subtitle_fetch(
|
||||
paths: &ArchivePaths,
|
||||
entry_uid: &str,
|
||||
options: SummaryBuildOptions, // Copy
|
||||
cookie_rules: &[database::CookieRule],
|
||||
transcription: Option<&transcriber::TranscriptionRequest>,
|
||||
) -> Result<SummaryInput>
|
||||
```
|
||||
|
||||
Body, in order:
|
||||
1. Call `subtitles::fetch_subtitles_for_entry(paths, entry_uid, cookie_rules)?`. It now returns a `SubtitleFetchOutcome` (§6.6). Log `eprintln!("info: summary {entry_uid}: subtitle fetch added {} artifact(s)", outcome.added)`.
|
||||
2. Match `build_summary_input(paths, entry_uid, options)`:
|
||||
- `Ok(input)`: return it.
|
||||
- `Err(e) if is_no_subtitles_error(&e) && transcription.is_some()`: continue to step 3.
|
||||
- `Err(e)`: return `Err(e)`. This is the old behaviour when no engine was requested.
|
||||
3. Call `transcriber::transcribe_entry(paths, entry_uid, request, outcome.original_language.as_deref(), cookie_rules)?`. It returns the number of artifact rows it inserted. Language refusal and failures come back as errors carrying a `TranscriptionUserMessage` (§8).
|
||||
4. Match `build_summary_input(paths, entry_uid, options)` once more:
|
||||
- `Ok(input)`: return it.
|
||||
- `Err(e) if is_no_subtitles_error(&e)`: return `Err(e.context(TranscriptionUserMessage(NO_SUBTITLES_AFTER_TRANSCRIPTION_MESSAGE.into())))`. In practice this only happens if a race left an unusable track, because `transcribe_entry` already rejects empty transcripts.
|
||||
- `Err(e)`: return `Err(e)`.
|
||||
|
||||
The server flow after this call is unchanged: `update_entry_summary_input_sha256` with the real digest, then `summarize_prebuilt_entry`. On error, `record_background_summary_failure`.
|
||||
|
||||
---
|
||||
|
||||
## 4. Engines
|
||||
|
||||
### 4.1 Comparison
|
||||
|
||||
| | Whisper (whisper.cpp / faster-whisper) | NVIDIA Parakeet TDT 0.6B (v2 / v3) | Fermion Research Phonon-2 |
|
||||
|---|---|---|---|
|
||||
| Languages | ~99 languages with multilingual models; `*.en` models are English-only | v2: English only. v3: 25 European languages with automatic language detection [INFERENCE: check the Hugging Face model card] | **English only** (vendor docs: "All of them transcribe English from 16 kHz audio") |
|
||||
| Model size | tiny ≈75 MB … large-v3 ≈3 GB; large-v3-turbo 1,618 MB (figure from Fermion's comparison table) | 2,508 MB at full precision (Fermion's table, v3); int8 ONNX builds are smaller | **164 MB** (≈2.1 bits per encoder weight) |
|
||||
| Accuracy (Open ASR Leaderboard, 7 English sets, avg WER, Fermion's table) | large-v3-turbo 6.58 % | v3: 4.96 % | 5.21 % |
|
||||
| Hardware | whisper.cpp: CPU (AVX/NEON), Apple Metal, CUDA/Vulkan builds. faster-whisper: CPU int8 or CUDA (CTranslate2) | NeMo: PyTorch, NVIDIA GPU recommended (CPU works but slowly) [INFERENCE]. parakeet-mlx: Apple silicon. sherpa-onnx: CPU int8 | Apple silicon GPU through MLX (174× realtime on an M5 MacBook Air); x86-64/Arm CPU engine in C (AVX-512 VNNI / AVX2 / NEON; 142.8× realtime on 8 Zen 5 cores); NVIDIA GPU (CUDA graphs); Windows CPU |
|
||||
| Runtime install | whisper.cpp: one native binary `whisper-cli` plus a ggml model file (nixpkgs `whisper-cpp` [INFERENCE: attribute and binary name in the pinned nixpkgs]). faster-whisper: `pip install faster-whisper` | `pip install nemo_toolkit[asr]` (heavy), `pip install parakeet-mlx`, or sherpa-onnx binaries [INFERENCE] | `pip install fermion-research` plus a platform runtime: on Apple silicon `pip install mlx mlx-audio mlx-lm soundfile scipy zstandard`; on Linux/Windows CPU `pip install fermion-research torch safetensors soundfile scipy zstandard` (CPU torch wheel). Containers: `ghcr.io/fermionresearch/phonon-cpu:2.0.6`, `ghcr.io/fermionresearch/phonon-cuda:1.0.5` |
|
||||
| Native output | whisper.cpp writes `.vtt`/`.srt`/`.json` directly. faster-whisper: Python API only (segments with `start`, `end`, `text`) | NeMo/parakeet-mlx: Python APIs with segment timestamps. No archivr-compatible CLI, so a wrapper script is needed | `fermion transcribe <model> <file>`: transcript-only stdout; `--json` gives text, model id, timings, per-segment start/end, a `words` list and a `truncated` flag. No VTT output |
|
||||
| Licence | whisper.cpp MIT; OpenAI Whisper weights MIT; faster-whisper and CTranslate2 MIT | Weights CC-BY-4.0 (attribution required on redistribution); NeMo Apache-2.0 [INFERENCE] | Weights **CC-BY-4.0** ("the licence of NVIDIA's Parakeet TDT 0.6B v3, from which they derive"). Phonon-1 models are Apache-2.0. The licence of the `fermion-research` CLI package is **[UNKNOWN]** (not stated on the pages read) |
|
||||
| Long audio | Handled internally (30 s windows) | NeMo full attention has a maximum single-pass length, roughly 24 min, so the wrapper must chunk or switch to local attention [INFERENCE] | Files longer than 35 s are decoded in 25–35 s windows cut at pauses and joined with single spaces |
|
||||
| Load cost per call | Model load is a few seconds | NeMo cold start is tens of seconds [INFERENCE] | "The first command in a session loads the engine (10 to 40 s)". Archivr starts one process per job, so every job pays this |
|
||||
|
||||
Sources: <https://www.fermionresearch.com/research/phonon-2/> and <https://www.fermionresearch.com/docs/speech/>, fetched 2026-10-05. Weights: <https://huggingface.co/FermionResearch/Phonon-2>. Everything not marked as coming from those pages is general knowledge and carries [INFERENCE] where it matters.
|
||||
|
||||
Archivr never redistributes weights, so attribution under CC-BY-4.0 is the duty of whoever installs or redistributes the model. If a future Docker image bundles Parakeet or Phonon-2 weights, it must carry attribution (§10).
|
||||
|
||||
### 4.2 The single output contract
|
||||
|
||||
Every engine adapter must end with **a VTT file inside the job's temp directory**, written by a subprocess or derived from one. Archivr then reads that file. This is the same rule as the codex provider: parse a file, never a free-form stdout. The one controlled exception is Phonon-2's `--json` stdout, which the vendor documents as clean ("Standard output carries only the transcript … Progress, warnings, and timings go to standard error"). The adapter parses it strictly and writes the VTT itself, so everything downstream still sees a file (§4.5).
|
||||
|
||||
### 4.3 Whisper (`whisper`)
|
||||
|
||||
Two backends, chosen by `ARCHIVR_WHISPER_BACKEND`:
|
||||
|
||||
**`whisper_cpp` (default).** Invoke whisper.cpp's CLI directly:
|
||||
|
||||
```
|
||||
<ARCHIVR_WHISPER_CLI> -m <ARCHIVR_WHISPER_MODEL> -f <job>/audio.wav -l <hint|auto> -ovtt -oj -of <job>/transcript -np
|
||||
```
|
||||
|
||||
- Outputs: `<job>/transcript.vtt` and `<job>/transcript.json`. `-of` takes the path *without* an extension. `-np` suppresses everything except results.
|
||||
- Language: use the language detected by Whisper from the `-oj` JSON, at `result.language` [INFERENCE: check this key against the pinned whisper.cpp], falling back to the hint, falling back to `und`.
|
||||
- These flags were stable in whisper.cpp for a long time. Pin them with an argument-builder test, and verify them against the installed version during the manual smoke test [INFERENCE].
|
||||
- Older builds name the binary `main` or `whisper-cpp`. `ARCHIVR_WHISPER_CLI` handles that.
|
||||
|
||||
**`script`.** `ARCHIVR_WHISPER_CLI` is a user-supplied wrapper that follows the **script contract** (§4.6), for example around faster-whisper (reference script in Appendix A.1).
|
||||
|
||||
Language hint (both backends): `h = language_base(original_language)`. Pass `h` only if it matches `^[a-z]{2}$` (Whisper's codes are mostly ISO 639-1). Otherwise pass `auto` (whisper.cpp) or omit `--language` (script). Never pass an unvalidated string.
|
||||
|
||||
### 4.4 NVIDIA Parakeet (`parakeet`)
|
||||
|
||||
Always uses the **script contract** (§4.6). Parakeet has no CLI that writes VTT and takes archivr's arguments. `ARCHIVR_PARAKEET_MODEL` (default `nvidia/parakeet-tdt-0.6b-v3`) is passed to the script as `--model`; the script decides how to load it (Hugging Face id, `.nemo` path, MLX repo, ONNX dir). Reference wrappers: Appendix A.2 (NeMo) and A.3 (parakeet-mlx).
|
||||
|
||||
Language support depends on the model, which archivr can't introspect. `ARCHIVR_PARAKEET_LANGUAGES` is an optional allowlist of base codes (§5); recommend `en` for v2. `--language` is passed when the hint is known, and the script may ignore it (v3 auto-detects).
|
||||
|
||||
### 4.5 Fermion Research Phonon-2 (`phonon2`)
|
||||
|
||||
**English only. This is hard-coded, not configurable.**
|
||||
|
||||
Invocation (from the vendor docs; the `--json` position follows their example):
|
||||
|
||||
```
|
||||
<ARCHIVR_PHONON2_CLI> transcribe <ARCHIVR_PHONON2_MODEL> <job>/audio.wav --json
|
||||
```
|
||||
|
||||
Defaults: CLI `fermion`, model `phonon-2`. Documented aliases are `phonon-2`, `phonon2`, `phonon`, `speech`, `stt`, `asr`. `phonon-1` and `phonon-1-micro` also work but are not the recommended model. The CLI also exposes `phonon transcribe <file>`; archivr uses the `fermion transcribe <model> <file>` form so the model is explicit.
|
||||
|
||||
- Input: the CLI reads anything libsndfile decodes (wav/flac/ogg/aiff) and **refuses mp3 and m4a**, printing an ffmpeg command. Archivr always passes the 16 kHz mono PCM WAV from §6.3, so this never happens.
|
||||
- Output: stdout is a single JSON object. The vendor describes these fields: the text; the model id; decode-only and wall-clock seconds; a start and end time per decoded segment; a `words` list with per-word start/end (Phonon-2 only); and a `truncated` flag. **The exact key names of the segment list and its members are [UNKNOWN].** Implementation steps:
|
||||
1. Run `fermion transcribe phonon-2 sample.wav --json` once on a real install.
|
||||
2. Paste the output (trimmed) as a test fixture const in `transcriber.rs`.
|
||||
3. Write `phonon_json_to_vtt` against those real keys.
|
||||
|
||||
Until a real sample confirms the shape, the parser should accept, in this order:
|
||||
- a top-level array of segment objects, each with numeric start/end seconds and a text string, under whichever key the sample shows (expected something like `segments`);
|
||||
- otherwise, the `words` list grouped into cues of at most 7 s or 84 characters, split at word boundaries;
|
||||
- otherwise, the top-level text as a single cue from `00:00:00.000` to the WAV duration. The WAV duration is `(file_len - 44) / 32000` seconds for 16 kHz mono s16le; §6.3 guarantees that format.
|
||||
- If `truncated` is `true`: `eprintln!("warn: phonon2 reported truncated segments for {entry_uid}")` and still accept the output.
|
||||
- Write the VTT to `<job>/transcript.vtt`. Cue timestamps are formatted `HH:MM:SS.mmm`. Cue text gets `&`, `<`, `>` escaped (`&`, `<`, `>`); the reducer decodes them again.
|
||||
- Language stored on the artifact: always `en`.
|
||||
- The CLI is a Python program. On first use it may download weights into `~/.cache` (the container examples mount `/home/phonon/.cache`), so the server user needs a writable `HOME` or cache directory (§10).
|
||||
- The **licence of the CLI package is [UNKNOWN]**. The weights are CC-BY-4.0.
|
||||
|
||||
### 4.6 Script contract (Whisper `script` backend, Parakeet)
|
||||
|
||||
Archivr runs:
|
||||
|
||||
```
|
||||
<executable> --input <job>/audio.wav --output <job>/transcript.vtt --model <model> [--language <xx>]
|
||||
```
|
||||
|
||||
The script must:
|
||||
- Exit 0 only after writing a WebVTT file to `--output`. That means a `WEBVTT` header, then cues `HH:MM:SS.mmm --> HH:MM:SS.mmm` followed by text lines, with blocks separated by blank lines.
|
||||
- Optionally write `<output>.lang` next to it, containing a single language code it detected (e.g. `de`). Archivr uses it only if it passes `is_safe_language_code`.
|
||||
- Treat `--language` as a hint it may ignore.
|
||||
- Send anything it prints to stdout or stderr. Archivr ignores stdout and keeps the last 4 KiB of stderr for its logs.
|
||||
- Write nothing outside `--output`'s directory except model caches.
|
||||
- Accept being killed with SIGKILL when the timeout expires.
|
||||
|
||||
The audio is already 16 kHz mono PCM WAV, so scripts never resample.
|
||||
|
||||
---
|
||||
|
||||
## 5. Configuration (env vars only, never TOML)
|
||||
|
||||
Resolution follows `provider_from_env`:
|
||||
- A missing required var produces an error naming that exact var (`required_env`).
|
||||
- Optional values use `env_or`/`optional_env`.
|
||||
- Timeouts use `env_timeout`.
|
||||
- CLIs that have a conventional install use `resolve_cli`: env override → well-known absolute paths → `$HOME/.local/bin/<bare>` → bare name on `PATH`.
|
||||
|
||||
Move these four private helpers from `summarizer.rs` into a new `crates/archivr-core/src/env_config.rs` as `pub(crate)` and update `summarizer.rs` to import them. That gives one convention with two users, not a copy.
|
||||
|
||||
| Variable | Default | Required when | Meaning |
|
||||
|---|---|---|---|
|
||||
| `ARCHIVR_TRANSCRIBE_ENGINES` | *(unset: feature off)* | always, to enable the feature | Comma-separated list of enabled engine kinds: `whisper`, `parakeet`, `phonon2`. Entries are trimmed and lowercased, empty entries are dropped, and duplicates are removed keeping the first. Unknown names get one `eprintln!("warn: …")` per call and are otherwise ignored. |
|
||||
| `ARCHIVR_WHISPER_CLI` | resolved: `/opt/homebrew/bin/whisper-cli`, `/usr/local/bin/whisper-cli`, `$HOME/.local/bin/whisper-cli`, `whisper-cli` | `whisper` enabled | whisper.cpp binary, or the wrapper script when the backend is `script`. With the `script` backend this var is **required** (`required_env`): auto-discovery would find whisper-cli, which does not follow the script contract. |
|
||||
| `ARCHIVR_WHISPER_MODEL` | — | `whisper` enabled | whisper.cpp: path to a ggml model file. Script: passed through as `--model` (e.g. `large-v3-turbo`). |
|
||||
| `ARCHIVR_WHISPER_BACKEND` | `whisper_cpp` | — | `whisper_cpp` or `script`. Any other value is an error naming the var and the allowed values. |
|
||||
| `ARCHIVR_WHISPER_LANGUAGES` | *(unset: any)* | — | Optional allowlist of base language codes (e.g. `en` for a `*.en` model). |
|
||||
| `ARCHIVR_PARAKEET_CLI` | — | `parakeet` enabled | Wrapper script following §4.6 (`required_env`). |
|
||||
| `ARCHIVR_PARAKEET_MODEL` | `nvidia/parakeet-tdt-0.6b-v3` | — | Passed as `--model`. |
|
||||
| `ARCHIVR_PARAKEET_LANGUAGES` | *(unset: any)* | — | Optional allowlist; recommend `en` for v2. |
|
||||
| `ARCHIVR_PHONON2_CLI` | resolved: `/opt/homebrew/bin/fermion`, `/usr/local/bin/fermion`, `$HOME/.local/bin/fermion`, `fermion` | — | The `fermion` CLI from `pip install fermion-research`. |
|
||||
| `ARCHIVR_PHONON2_MODEL` | `phonon-2` | — | Model name or alias passed to `fermion transcribe`. |
|
||||
| `ARCHIVR_TRANSCRIBE_TIMEOUT` | `3600` | — | Seconds of wall-clock budget for one transcription job (audio acquisition, ffmpeg and engine together; §8.2). |
|
||||
| `ARCHIVR_FFMPEG` | `ffmpeg` | — | ffmpeg binary (`env_or`). Set by the Nix wrappers and the Dockerfile (§10). |
|
||||
|
||||
Notes:
|
||||
- `ARCHIVR_TRANSCRIBE_ENGINES` is the **gate**. An engine whose vars are complete but which is not listed there is not offered and is rejected at POST. This keeps a half-configured host from accidentally exposing a CPU-heavy feature.
|
||||
- Language allowlists hold base codes compared with `language_base`. Parsing is the same as for the engine list. An allowlist that ends up empty is the same as unset.
|
||||
- No secrets are involved. Engine vars are paths and model names, so nothing needs a secret file. NixOS users can still use `environmentFile` (§10).
|
||||
|
||||
---
|
||||
|
||||
## 6. Core design
|
||||
|
||||
### 6.1 New module `crates/archivr-core/src/transcriber.rs`
|
||||
|
||||
Add `pub mod transcriber;` (and `pub(crate) mod env_config;`, `pub(crate) mod process;`) to `lib.rs`. Everything is synchronous.
|
||||
|
||||
```rust
|
||||
pub const TRANSCRIBE_ENGINE_KINDS: [&str; 3] = ["whisper", "parakeet", "phonon2"];
|
||||
pub const DEFAULT_TRANSCRIBE_TIMEOUT_SECS: u64 = 3600;
|
||||
pub const TRANSCRIBE_SAMPLE_RATE_HZ: u32 = 16_000;
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum WhisperBackend { WhisperCpp, Script }
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct TranscriberConfig {
|
||||
pub kind: &'static str, // one of TRANSCRIBE_ENGINE_KINDS
|
||||
pub executable: PathBuf,
|
||||
pub model: String,
|
||||
pub whisper_backend: WhisperBackend, // ignored unless kind == "whisper"
|
||||
pub languages: Option<Vec<String>>, // base codes; None = any. phonon2: always Some(["en"])
|
||||
pub timeout_secs: u64,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct TranscriptionSettings {
|
||||
pub ffmpeg: PathBuf,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq, serde::Serialize)]
|
||||
pub struct TranscriberInfo {
|
||||
pub kind: &'static str,
|
||||
pub label: &'static str, // "Whisper" | "NVIDIA Parakeet" | "Phonon-2"
|
||||
pub english_only: bool, // languages == Some(["en"])
|
||||
pub languages: Option<Vec<String>>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct TranscriptOutput {
|
||||
pub vtt_path: PathBuf, // inside out_dir
|
||||
pub language: Option<String>, // detected/assumed; validated with is_safe_language_code
|
||||
}
|
||||
|
||||
/// `Send + Sync` so a boxed transcriber can cross into the server's `spawn_blocking` worker.
|
||||
pub trait Transcriber: Send + Sync {
|
||||
fn kind(&self) -> &'static str;
|
||||
fn label(&self) -> &'static str;
|
||||
fn model(&self) -> &str;
|
||||
fn timeout_secs(&self) -> u64;
|
||||
/// `None` = language unknown → always true (see §6.5).
|
||||
fn supports_language(&self, original_language: Option<&str>) -> bool;
|
||||
fn supported_languages(&self) -> Option<&[String]>;
|
||||
fn transcribe(&self, audio_wav: &Path, lang_hint: Option<&str>, out_dir: &Path,
|
||||
deadline: std::time::Instant) -> Result<TranscriptOutput>;
|
||||
}
|
||||
|
||||
pub struct TranscriptionRequest {
|
||||
pub transcriber: Box<dyn Transcriber>,
|
||||
pub settings: TranscriptionSettings,
|
||||
}
|
||||
|
||||
pub fn enabled_engine_kinds() -> Vec<&'static str>; // parses ARCHIVR_TRANSCRIBE_ENGINES
|
||||
pub fn transcriber_from_env(kind: &str) -> Result<TranscriberConfig>; // unknown kind → "unknown transcription engine: x (expected one of whisper, parakeet, phonon2)"
|
||||
pub fn transcriber_from_config(cfg: TranscriberConfig) -> Box<dyn Transcriber>;
|
||||
pub fn transcription_settings_from_env() -> TranscriptionSettings;
|
||||
/// Enabled AND configured engines, in TRANSCRIBE_ENGINE_KINDS order. A configuration error is logged once per call and the engine is left out.
|
||||
pub fn available_transcribers() -> Vec<TranscriberInfo>;
|
||||
/// Server entry point: kind must be enabled (else error naming ARCHIVR_TRANSCRIBE_ENGINES) and configured (else the transcriber_from_env error).
|
||||
pub fn request_from_env(kind: &str) -> Result<TranscriptionRequest>;
|
||||
pub fn transcribe_entry(paths: &ArchivePaths, entry_uid: &str, request: &TranscriptionRequest,
|
||||
original_language: Option<&str>, cookie_rules: &[database::CookieRule]) -> Result<usize>;
|
||||
```
|
||||
|
||||
Notes on the plan's outline:
|
||||
- The plan sketched `transcribe(..) -> Result<PathBuf>`. This spec returns `TranscriptOutput` so the detected language can be stored, and adds an explicit `deadline` so one budget covers all steps.
|
||||
- Implement one private struct per engine: `WhisperCppTranscriber`, `ScriptTranscriber` (for both Whisper `script` and Parakeet; it carries `kind`/`label`), and `Phonon2Transcriber`. `transcriber_from_config` boxes the matching one.
|
||||
- The trait makes in-process fakes possible in tests (§11).
|
||||
|
||||
Pure helpers. All are private unless noted, and all are unit-tested:
|
||||
|
||||
```rust
|
||||
fn whisper_cpp_args(model: &str, wav: &Path, lang_hint: Option<&str>, out_prefix: &Path) -> Vec<OsString>;
|
||||
fn script_args(wav: &Path, out_vtt: &Path, model: &str, lang_hint: Option<&str>) -> Vec<OsString>;
|
||||
fn phonon2_args(model: &str, wav: &Path) -> Vec<OsString>;
|
||||
fn ffmpeg_resample_args(input: &Path, out_wav: &Path) -> Vec<OsString>;
|
||||
fn whisper_language_hint(original_language: Option<&str>) -> Option<String>; // ^[a-z]{2}$ after language_base
|
||||
fn phonon_json_to_vtt(json: &str, wav_duration_secs: f64) -> Result<String>;
|
||||
fn format_vtt_timestamp(seconds: f64) -> String; // "HH:MM:SS.mmm", clamps negatives to 0
|
||||
fn wav_duration_secs(byte_len: u64) -> f64; // (len - 44).max(0) / 32000
|
||||
fn parse_language_list(raw: Option<&str>) -> Option<Vec<String>>;
|
||||
fn select_audio_source(store_path: &Path, primary: &[database::RoleArtifact]) -> Option<PathBuf>;
|
||||
```
|
||||
|
||||
### 6.2 `transcribe_entry`, step by step
|
||||
|
||||
1. **Lock** (§6.7). Acquire the process-wide transcription slot, waiting at most `timeout_secs`. After the wait, start the job clock with `deadline = Instant::now() + timeout_secs`. Waiting in the queue does not use up the job budget.
|
||||
2. **Re-check** (concurrency, mirrors D6). Open the DB and call `entry_source_info(conn, entry_uid)`; a missing entry is `bail!("entry not found: {entry_uid}")`. If it is not `youtube`/`video`, `bail!` (defensive; the caller already gated). If `list_entry_artifacts_by_role(conn, entry_id, SUBTITLE_ARTIFACT_ROLE)` now holds any artifact whose `subtitle_to_transcript` is non-empty, another request has already produced a track. Return `Ok(0)` without transcribing.
|
||||
3. **Language gate** (§6.5): if `!transcriber.supports_language(original_language)`, return the language-unsupported error (§8.1).
|
||||
4. **Job dir**: `job = store_path/temp/transcribe-<Uuid::new_v4().simple()>`. Wrap it in a private `TempDirGuard(PathBuf)` whose `Drop` runs `let _ = fs::remove_dir_all(..)`, so cleanup also happens on `?` and on panics.
|
||||
5. **Audio source** (§6.3): an archived media file, or failing that a yt-dlp audio-only download into `job`.
|
||||
6. **Resample** with ffmpeg into `job/audio.wav` (§6.3).
|
||||
7. **Transcribe**: `transcriber.transcribe(&job.join("audio.wav"), hint, &job, deadline)`. The hint is `original_language` (each adapter derives its own form).
|
||||
8. **Validate**: `vtt_path` must exist and be non-empty. `subtitle_to_transcript(read_to_string(vtt_path)?)` must be non-empty; otherwise the job ends with the no-speech error (§8.1).
|
||||
9. **Stage and archive**:
|
||||
- Build `StagedSubtitle { path: vtt_path, language, kind: SubtitleKind::Transcribed, format: "vtt".into(), original_language: original_language.map(str::to_string) }`. `language` is the first of `output.language` (if safe), `original_language` (if safe), then `"und"`. For `phonon2` it is always `"en"`.
|
||||
- Call `subtitles::archive_staged_subtitles(store_path, vec![staged])`. An empty result means the move failed, which is an error.
|
||||
10. **Register**: `subtitles::register_transcript_artifact(conn, store_path, entry_id, &archived, transcriber.kind(), transcriber.model())` (§7). Log `eprintln!("info: transcribed {entry_uid} with {kind} ({model}) in {secs:.1}s")` and return the inserted count.
|
||||
11. The guard drops and removes `job`. The audio WAV and any yt-dlp audio are never archived.
|
||||
|
||||
### 6.3 Audio acquisition and resampling
|
||||
|
||||
**Source selection** (`select_audio_source`). Go through `list_entry_artifacts_by_role(conn, entry_id, "primary_media")` in id order. Take the first artifact where both hold:
|
||||
- the extension (lowercased, from `relpath`) is in `{mp4, m4a, webm, mkv, mov, mp3, opus, ogg, oga, flac, wav, aac}`, **or** the MIME type starts with `audio/` or `video/`;
|
||||
- `store_path.join(relpath).is_file()`.
|
||||
|
||||
YouTube captures always archive media this way: an mp4 for video qualities, or the extracted audio file for the `audio` quality. So the archived file is the normal source, and it costs no network and no YouTube request.
|
||||
|
||||
**Fallback: yt-dlp audio-only download.** Used only when there is no usable archived file (pruned store, legacy entry) **and** `canonical_url` is `http(s)://`, the same gate as `fetch_subtitles_for_entry`. That gate also keeps tests from spawning yt-dlp. Add this to `downloader/ytdlp.rs`:
|
||||
|
||||
```rust
|
||||
pub fn download_audio_for_transcription(url: &str, store_path: &Path, stage_key: &str,
|
||||
cookies: &HashMap<String, String>, timeout_secs: u64) -> Result<PathBuf>;
|
||||
fn audio_only_args(url: &str, cookie_file: Option<&Path>, out_template: &Path) -> Vec<OsString>;
|
||||
// url, -f, bestaudio/best, --no-playlist, [--cookies f], -o temp/<key>/<key>.audio.%(ext)s
|
||||
```
|
||||
|
||||
- Build it with `yt_dlp_command(&resolve_yt_dlp())` and spawn it through `process::run_with_timeout` (§6.4), with the remaining budget.
|
||||
- Use the same UUID-named cookie-file pattern and cleanup as `download` (`capture::resolve_cookies_for_url(cookie_rules, url)`).
|
||||
- **No `-x`**: archivr's ffmpeg step converts, and `-x` would make yt-dlp call ffmpeg a second time.
|
||||
- The stage key is the job dir's name, so the guard cleans it up.
|
||||
- Return the single non-`.part`/`.ytdl` file matching `<key>.audio.*`, or bail.
|
||||
- The fetched audio is transient and **not** archived: the entry already has its own media record, and a second media artifact would distort `cached_bytes` and the entry view.
|
||||
|
||||
If there is no source at all, fail with the no-audio copy (§8.1).
|
||||
|
||||
**Resampling.** Always run ffmpeg, even when the input is already WAV, so every engine sees exactly one format:
|
||||
|
||||
```
|
||||
<ARCHIVR_FFMPEG> -nostdin -hide_banner -loglevel error -y -i <input> -map 0:a:0 -vn -sn -dn -ac 1 -ar 16000 -c:a pcm_s16le <job>/audio.wav
|
||||
```
|
||||
|
||||
- `-map 0:a:0` makes ffmpeg fail when there is no audio stream. Map that failure to the audio-extraction copy (§8.1).
|
||||
- The output is 16 kHz mono signed 16-bit PCM WAV, 32,000 bytes per second: about 115 MB per hour of audio in `store/temp/`. Document the disk requirement in the README.
|
||||
- Run through `process::run_with_timeout` with the remaining budget. On non-zero exit, log the stderr tail (`eprintln!`) and return the sanitized copy.
|
||||
|
||||
### 6.4 Subprocess runner with timeout: `crates/archivr-core/src/process.rs`
|
||||
|
||||
The repo deliberately has no `wait_timeout` dependency. The existing `summarizer::run_cli` enforces timeouts with a thread, a channel and `recv_timeout`. It has a latent flaw for long jobs: it reads **stderr only after the child exits**. whisper.cpp, NeMo and yt-dlp write a lot to stderr. Once the pipe buffer (~64 KiB) fills, the child blocks, and the run ends in a spurious timeout.
|
||||
|
||||
Fix it once and share it:
|
||||
|
||||
```rust
|
||||
pub(crate) struct ProcessOutput { pub stdout: String, pub stderr_tail: String /* last 4 KiB, lossy UTF-8 */ }
|
||||
|
||||
/// Spawns `executable args…`, optionally writes `stdin`, drains stdout and stderr on their own threads,
|
||||
/// and kills the child if it is still running at `timeout`. Non-zero exit → Err("{exe} exited with {status}: {truncated stderr}").
|
||||
/// Timeout → Err("{exe} timed out after {secs}s").
|
||||
pub(crate) fn run_with_timeout(executable: &Path, args: &[OsString], stdin: Option<&str>, timeout: Duration) -> Result<ProcessOutput>;
|
||||
```
|
||||
|
||||
- Move the body of `run_cli` here and add a stderr-draining thread that keeps a bounded tail.
|
||||
- `summarizer::run_cli` becomes a thin wrapper (`args` mapped to `OsString`, `Some(prompt)`, timeout), so provider behaviour is unchanged. Keep its existing tests (`run_cli_round_trips_stdin_to_stdout`, `run_cli_kills_a_child_that_overruns_its_timeout`, `run_cli_reports_a_nonzero_exit`, `codex_positional_fallback_honors_cli_timeout`).
|
||||
- Timeout errors must be recognisable without string matching. Attach a `ProcessTimedOut { secs }` sentinel (Display `"timed out after {secs}s"`) with `.context(..)`, and detect it via `chain().any(downcast_ref)`, the pattern `UnsupportedSummaryContent` already uses.
|
||||
- Timeout for each step: `deadline.saturating_duration_since(Instant::now())`. If that is zero, fail with the timeout copy before spawning anything.
|
||||
- **Deviation (implementation): process-group kill.** "Kills the child" alone leaves grandchildren alive: a `script` backend that runs `python …` or `nemo` without `exec`, or yt-dlp's ffmpeg, kept running after a timeout and held the output pipes open. On unix the runner now spawns the child in its own process group (`CommandExt::process_group(0)`) and on timeout SIGKILLs the whole group via `libc::kill(-pgid, SIGKILL)` (unix-only `libc` dependency; never for pid ≤ 1; `ESRCH` ignored) before reaping. Shelling out to `kill -KILL -- -<pid>` was dropped because the Debian slim runtime image ships no `kill` binary, so the group kill silently did nothing there. After the direct child exits normally, the pipe readers get a 2 s grace (capped by the remaining budget); if a grandchild still holds the pipes, the group is killed and the readers get one more grace. yt-dlp's private `run_with_timeout` in `downloader/ytdlp.rs` follows the same rules. This also tightens §4.6: script backends' subprocesses are killed with them.
|
||||
|
||||
### 6.5 Language gating (Phonon-2 English-only, optional allowlists)
|
||||
|
||||
`supports_language(original_language)`:
|
||||
- `languages == None`: `true`.
|
||||
- `original_language == None` (unknown): **`true`**. Many YouTube videos have no `language` field in yt-dlp metadata [INFERENCE]. A user who picks an English-only engine for such a video has made an explicit choice, and refusing would make Phonon-2 unusable in that case. Log `eprintln!("warn: {kind}: original language unknown for {entry_uid}; assuming it is supported")`.
|
||||
- Otherwise: `languages.contains(&language_base(original_language))`. `en`, `en-US`, `en-GB` and `en-orig` all pass for Phonon-2; `de`, `de-orig` and `pt-BR` are refused.
|
||||
|
||||
`phonon2` is constructed with `languages: Some(vec!["en".into()])` regardless of env, so it can't be configured away.
|
||||
|
||||
Promote `ytdlp::language_base` and `ytdlp::is_safe_language_code` to `pub` and reuse them. Do not copy them into `transcriber.rs`.
|
||||
|
||||
### 6.6 Knowing the original language at summary time
|
||||
|
||||
The `--dump-json` metadata is **not persisted** on the entry (capture only derives the title from it). The fallback therefore takes the language from the probe that `fetch_subtitles_for_entry` already makes. Change its return type (clean cutover; its only caller is `build_summary_input_with_subtitle_fetch`):
|
||||
|
||||
```rust
|
||||
#[derive(Debug, Clone, Default, PartialEq, Eq)]
|
||||
pub struct SubtitleFetchOutcome {
|
||||
pub added: usize,
|
||||
pub original_language: Option<String>,
|
||||
}
|
||||
pub fn fetch_subtitles_for_entry(paths: &ArchivePaths, entry_uid: &str,
|
||||
cookie_rules: &[database::CookieRule]) -> Result<SubtitleFetchOutcome>;
|
||||
```
|
||||
|
||||
- Factor the original-language derivation out of `plan_subtitle_request` into `pub fn original_language_from_metadata(value: &serde_json::Value) -> Option<String>` in `ytdlp.rs`: the `language` field (trimmed, non-empty), otherwise the first sorted safe `automatic_captions` key ending in `-orig` with the suffix removed. `plan_subtitle_request` calls it.
|
||||
- In `fetch_subtitles_for_entry`, set `original_language` from `original_language_from_metadata` **right after `fetch_metadata` succeeds**, *before* the `plan_subtitle_request(..) == None` early return. A video with no captions is exactly the case where planning returns `None`, and the language must survive it.
|
||||
- Right after `entry_source_info`, compute `existing_original_language`: the first existing `subtitle` artifact (id order) whose `parse_subtitle_metadata(..).original_language` is `Some`. This is one cheap query.
|
||||
- **Every** early return carries it: entry not `youtube`/`video`, non-`http(s)` URL, existing subtitle artifacts (the S2 step-3 re-check), unreachable video, and no plannable tracks. A value from a successful probe overrides it.
|
||||
- `added` counts only rows inserted by this call. The re-check early return therefore reports `added: 0`, not the number of existing artifacts as the S2 algorithm's `Ok(len)` did. The only caller just logs the count.
|
||||
|
||||
### 6.7 Concurrency
|
||||
|
||||
- **One transcription at a time per server process.** Engines saturate the CPU or GPU, and two parallel Whisper large runs can exhaust memory. Implement this as a private `static SLOT: (Mutex<bool>, Condvar)` in `transcriber.rs`, and acquire it with `Condvar::wait_timeout_while(guard, timeout, |busy| *busy)`.
|
||||
- If the wait times out, fail with the busy copy (§8.1).
|
||||
- A small RAII guard releases the slot and calls `notify_one` on drop.
|
||||
- The CLI never transcribes (summaries are server-only), so per-process is per-server.
|
||||
- **Same entry, concurrent requests.** The second request waits for the slot. Its re-check (§6.2 step 2) then finds the first request's transcribed track and returns `Ok(0)`, so the rebuild succeeds without a second transcription.
|
||||
- **Dedup.** `register_transcript_artifact` uses the same IMMEDIATE transaction and the `entry_has_artifact_blob` check as `register_subtitle_artifacts`. Identical VTT bytes never create two rows. Two runs that produce different bytes cannot happen, because the re-check runs under the slot.
|
||||
- **Server restart mid-job.** `fail_stalled_entry_summaries` already fails the pending row at startup. The job dir `temp/transcribe-*` is left behind, which matches how interrupted captures leave `temp/<timestamp>` today (§12).
|
||||
- Tokio's blocking pool holds one thread per waiting or running job. Each job is a `spawn_blocking`, as provider calls already are.
|
||||
|
||||
---
|
||||
|
||||
## 7. Artifact storage
|
||||
|
||||
- **Role: `subtitle`, not a new `transcript` role.** With `subtitle`, `youtube_transcript_content` already lists, ranks, reduces and labels the track. Registration, dedup, `cached_bytes` refresh and the fetch re-check all work unchanged. A separate role would need a second candidate list in the summarizer and a second "has text" check in `fetch_subtitles_for_entry`. The trade-off is accepted: once a transcribed track exists, the re-check in `fetch_subtitles_for_entry` treats the entry as having subtitles and no longer contacts YouTube (§12).
|
||||
- **New kind.** Add `SubtitleKind::Transcribed` in `ytdlp.rs`: `as_str() == "transcribed"`, and `parse("transcribed") == Transcribed`. Update every exhaustive `match` on `SubtitleKind`. yt-dlp staging never produces this kind.
|
||||
- **New origin.** `pub const SUBTITLE_ORIGIN_TRANSCRIPTION: &str = "transcription";` in `subtitles.rs`.
|
||||
- **Storage.** `storage_area = "raw"`, `blob_id = Some`, `logical_path = None`, blob MIME `text/vtt`, extension `vtt`. All of this comes for free from `archive_staged_subtitles` and the existing `BlobRecord` construction.
|
||||
- **metadata_json**:
|
||||
```json
|
||||
{"language":"en","kind":"transcribed","format":"vtt","original_language":"en"|null,
|
||||
"origin":"transcription","engine":"phonon2","model":"phonon-2"}
|
||||
```
|
||||
`model` is the configured value. If it is a filesystem path (it contains `/` or `\`), store only the file name, so no host path is persisted in the archive.
|
||||
- **Registration API.** Refactor the insertion loop of `register_subtitle_artifacts` into a private `insert_subtitle_rows(conn, store_path, entry_id, rows: &[(&ArchivedSubtitle, serde_json::Value)]) -> Result<usize>` that owns the transaction, the dedup check, the commit and `refresh_entry_cached_bytes`. Then:
|
||||
- `register_subtitle_artifacts(.., origin)` builds the five-key metadata and calls it (behaviour unchanged).
|
||||
- New `pub fn register_transcript_artifact(conn, store_path, entry_id, sub: &ArchivedSubtitle, engine: &str, model: &str) -> Result<usize>` adds `engine`/`model` with origin `SUBTITLE_ORIGIN_TRANSCRIPTION`.
|
||||
- **Ranking.** Transcribed tracks go below every manual track and above auto captions. `subtitle_track_rank` becomes:
|
||||
|
||||
| Rank | Track |
|
||||
|---|---|
|
||||
| 0 | manual + en |
|
||||
| 1 | manual + orig |
|
||||
| 2 | other manual |
|
||||
| 3 | **transcribed** (any language) |
|
||||
| 4 | auto/unknown + orig |
|
||||
| 5 | auto/unknown + en |
|
||||
| 6 | anything else |
|
||||
|
||||
Update the existing rank test's expected numbers. A transcribed track normally only exists when nothing else did, so the position only matters if subtitles appear later.
|
||||
- **Summary label.** No special case: `Transcript ({language}, transcribed subtitles):`, from the existing format string with `kind.as_str()`.
|
||||
- **Digest.** The content changes, so `input_sha256` changes and no cached subtitle-less row is ever reused. `PROMPT_VERSION` is not bumped.
|
||||
|
||||
---
|
||||
|
||||
## 8. Error handling, timeouts and copy
|
||||
|
||||
### 8.1 User-visible copy
|
||||
|
||||
Add a sentinel to `transcriber.rs`, following the `UnsupportedSummaryContent` pattern:
|
||||
|
||||
```rust
|
||||
#[derive(Debug)]
|
||||
pub struct TranscriptionUserMessage(pub String); // Display = the String
|
||||
pub fn transcription_user_message(error: &anyhow::Error) -> Option<String>; // first in chain()
|
||||
```
|
||||
|
||||
Every failure in `transcribe_entry` is built as `Err(anyhow!(<detailed diagnostic>).context(TranscriptionUserMessage(<copy>)))`:
|
||||
- The diagnostic (paths, exit status, stderr tail) goes to `eprintln!("warn: transcription {entry_uid}: {e:#}")`.
|
||||
- Only the copy reaches the row.
|
||||
|
||||
Error text is visible to authenticated users only; public readers never see diagnostics. Even so, host paths and engine stderr do not belong in the archive DB.
|
||||
|
||||
In `routes.rs`, `summary_failure_error_text` checks in this order:
|
||||
1. `transcriber::transcription_user_message(error)`
|
||||
2. `is_no_subtitles_error` → `NO_SUBTITLES_SUMMARY_MESSAGE`
|
||||
3. `is_unsupported_summary_content_error` → `UNSUPPORTED_SUMMARY_CONTENT_MESSAGE`
|
||||
4. otherwise `format!("{error:#}")`
|
||||
|
||||
Copy uses one line, curly apostrophes like the existing constants, and `{label}` from `Transcriber::label()`:
|
||||
|
||||
| Case | Copy |
|
||||
|---|---|
|
||||
| Language refused | `This video can’t be transcribed with {label} because it only supports {supported}, and the video’s original language is “{lang}”. Choose a different transcription engine.` Here `{supported}` is `English` for `["en"]`, otherwise `these languages: en, de, …`. Put this in a `pub fn transcription_language_unsupported_message(label, lang, supported: &[String]) -> String`. |
|
||||
| No audio source | `Local transcription with {label} couldn’t start: the archived media file is missing and the original video couldn’t be downloaded.` |
|
||||
| ffmpeg failed / no audio track | `Local transcription with {label} failed: the audio couldn’t be extracted from this video (it may have no audio track).` |
|
||||
| Engine non-zero exit | `Local transcription with {label} failed: the transcription engine exited with an error. Check the server log for details.` |
|
||||
| Engine wrote no or empty VTT, or unparsable Phonon JSON | `Local transcription with {label} failed: the transcription engine produced no subtitle file. Check the server log for details.` |
|
||||
| Budget exceeded (`ProcessTimedOut` in the chain, or a zero remaining budget) | `Local transcription with {label} timed out after {timeout_secs} seconds. Raise ARCHIVR_TRANSCRIBE_TIMEOUT or choose a faster engine.` |
|
||||
| Slot wait timed out | `Local transcription with {label} didn’t start because another transcription was still running. Try again later.` |
|
||||
| Transcript empty after reduction (silence or music) | `const NO_SUBTITLES_AFTER_TRANSCRIPTION_MESSAGE: &str = "This video can’t be summarized because no subtitles are available and local transcription found no speech in its audio.";` (in `summarizer.rs`) |
|
||||
|
||||
`NO_SUBTITLES_SUMMARY_MESSAGE` stays exactly as it is for requests that did not ask for transcription. When an engine was tried, every outcome uses one of the transcription-specific messages above, so the original copy never wrongly claims that nothing else was attempted.
|
||||
|
||||
### 8.2 Timeouts
|
||||
- **One budget per job**: `ARCHIVR_TRANSCRIBE_TIMEOUT` (default 3600 s), stored as `TranscriberConfig::timeout_secs`. It covers the yt-dlp audio fallback, ffmpeg and the engine. Every subprocess gets the remaining time; on expiry the child is killed (`process::run_with_timeout`).
|
||||
- Waiting for the slot has its own cap of the same value, and that wait is not counted in the job budget.
|
||||
- The summary provider's timeout (`ARCHIVR_SUMMARY_*_TIMEOUT`) applies only afterwards, to the LLM call. The two are independent.
|
||||
- A full hour is a realistic need on CPU-only hosts for Whisper large (whisper.cpp large-v3-turbo with Metal measured 17× realtime in Fermion's table; CPU-only runs are much slower [INFERENCE]). Phonon-2 on CPU handles an hour of audio in tens of seconds, plus 10–40 s of engine load.
|
||||
|
||||
### 8.3 Preflight (synchronous 400s, no row created)
|
||||
In `request_entry_summary_handler`, right after `provider_from_env` succeeds and before the preflight `spawn_blocking`:
|
||||
- `transcribe_engine` that is `Some` with trimmed non-empty `k` → `transcriber::request_from_env(k)`. On `Err`, return `ApiError::bad_request(&format!("{e:#}"))`. The message names the missing var or says `transcription engine 'k' is not enabled (add it to ARCHIVR_TRANSCRIBE_ENGINES)`.
|
||||
- An empty string is treated as absent.
|
||||
- Validation happens even if the entry later turns out to have subtitles. A configuration error is surfaced consistently and is cheap; the request is then simply never used.
|
||||
|
||||
### 8.4 Cleanup guarantees
|
||||
`TempDirGuard` removes `temp/transcribe-<uuid>` (WAV, engine outputs, yt-dlp audio, cookie file) on success, error and panic. The archived VTT has already been moved to `raw/` before the guard drops.
|
||||
|
||||
---
|
||||
|
||||
## 9. API and UI
|
||||
|
||||
### 9.1 Server (`crates/archivr-server/src/routes.rs`)
|
||||
- `SummaryRequestBody` gains `#[serde(default)] transcribe_engine: Option<String>`.
|
||||
- Update the handler doc comment: the body may carry `transcribe_engine`, which is used only for YouTube videos without subtitles and runs in the background.
|
||||
- After the preflight validation (§8.3), hold `transcription: Option<transcriber::TranscriptionRequest>`.
|
||||
- Move it into the background closure. Only the `FetchSubtitles` branch uses it: it calls `summarizer::build_summary_input_with_subtitle_fetch(&paths, &entry_uid, summary_options, &cookie_rules, transcription.as_ref())`.
|
||||
- The `Pending` (input already built) branch drops it unused. The 202 body is unchanged.
|
||||
- New route `.route("/api/summary/transcription-engines", get(transcription_engines_handler))`:
|
||||
- `auth_user.require_role(ROLE_USER)?`; guests and public readers get the existing 401/403 behaviour.
|
||||
- Returns `Json(transcriber::available_transcribers())`, e.g. `[{"kind":"phonon2","label":"Phonon-2","english_only":true,"languages":["en"]}]`.
|
||||
- It reads env only and spawns nothing, so it is cheap enough to call once per ContextRail mount.
|
||||
- An empty array means the feature is off.
|
||||
|
||||
### 9.2 Frontend
|
||||
- `frontend/src/api.js`:
|
||||
- `export async function fetchTranscriptionEngines({ signal } = {})` returns `getJson('/api/summary/transcription-engines', { signal })`. On a 401/403 the caller treats the result as `[]`.
|
||||
- `requestEntrySummary(archiveId, entryUid, { provider, force, includeImages, transcribeEngine, signal })` adds `transcribe_engine: transcribeEngine` to the JSON body **only when it is a non-empty string**. Extend the comment.
|
||||
- `frontend/src/components/ContextRail.jsx`:
|
||||
- State: `transcriptionEngines` (default `[]`), loaded once on mount when `!isPublicSession`; errors become `[]`. `transcribeEngine` is initialised from `sessionStorage['archivr:summary:transcribe-engine']` (const `SUMMARY_TRANSCRIBE_ENGINE_KEY`), default `''`, in the same try/catch style as `SUMMARY_PROVIDER_KEY`. Once engines load, reset to `''` if the stored value is not among them.
|
||||
- Render inside `.rail-summary-controls`, after the provider `<select>`, **only when** `transcriptionEngines.length > 0 && detail.summary.source_kind === 'youtube' && detail.summary.entity_kind === 'video'`:
|
||||
```jsx
|
||||
<select className="rail-summary-select" value={transcribeEngine}
|
||||
onChange={e => handleTranscribeEngineChange(e.target.value)}
|
||||
aria-label="Local transcription if no subtitles">
|
||||
<option value="">No local transcription</option>
|
||||
{transcriptionEngines.map(t => (
|
||||
<option key={t.kind} value={t.kind}>{t.label}{t.english_only ? ' (English only)' : ''}</option>
|
||||
))}
|
||||
</select>
|
||||
<p className="rail-summary-transcribe-note">Used only if this video has no subtitles. Transcription runs on this server and can take several minutes.</p>
|
||||
```
|
||||
- `handleTranscribeEngineChange` sets state and persists to sessionStorage (try/catch for private mode).
|
||||
- Pass `transcribeEngine` to `requestEntrySummary`. Keep the polling and generate callbacks scoped to the selected entry, as `AGENTS.md` requires.
|
||||
- Progress: the existing `running` / "Generating…" spinner and 1500 ms polling cover the whole fetch, transcribe and summarize sequence, because the row stays `pending` throughout. No new status values.
|
||||
- Errors: transcription copy arrives as the failed attempt's `error_text` and renders in the existing `rail-summary-error` paragraph. No logic change.
|
||||
- `frontend/src/styles.css`: `.rail-summary-transcribe-note`, styled like `.rail-summary-image-option__note` (small muted text). Use existing custom properties only.
|
||||
- **CaptureDialog: no change.** Capture-time transcription is a non-goal (§1). If it is added later, the control belongs next to the "Download subtitles" toggle as a `transcribe_engine` capture extension, with core support in `CaptureConfig`.
|
||||
|
||||
---
|
||||
|
||||
## 10. Deployment
|
||||
|
||||
**Nix (`flake.nix`)**:
|
||||
- Add `--set ARCHIVR_FFMPEG ${pkgs.ffmpeg}/bin/ffmpeg` to **both** the `archivr` and `archivr-server` `makeWrapper` calls, following the existing `--set` pattern. Today ffmpeg is only on the PATH of the `ytDlp` wrapper, not of the server.
|
||||
- Add `pkgs.ffmpeg` to the dev shell `buildInputs`.
|
||||
- Optionally add `pkgs.whisper-cpp` to the dev shell for local testing [INFERENCE: check the attribute name and that it ships `whisper-cli` in the pinned `nixos-unstable`].
|
||||
- Do **not** wrap any engine or model into the packages. Engines and models stay user-supplied.
|
||||
|
||||
**NixOS module (`modules/nixos/archivr-server.nix`)**. The module has no way to pass env vars today. Add:
|
||||
- `environment = lib.mkOption { type = lib.types.attrsOf lib.types.str; default = { }; description = "Extra environment variables (e.g. ARCHIVR_TRANSCRIBE_ENGINES, ARCHIVR_PHONON2_CLI, LLM provider settings)."; }`, mapped to `systemd.services.archivr-server.environment`.
|
||||
- `environmentFile = lib.mkOption { type = lib.types.nullOr lib.types.path; default = null; }`, mapped to `serviceConfig.EnvironmentFile` when non-null. This is useful for the LLM API keys too.
|
||||
- Hardening impact:
|
||||
- `ProtectSystem = "strict"` keeps model files readable but read-only, which is fine.
|
||||
- Python engines that download weights on first use need a writable cache. Set `HOME=/var/lib/archivr-server`, already the user's home and inside `StateDirectory`, plus `XDG_CACHE_HOME`/`HF_HOME` under it in the module's default `environment`. Use `lib.mkDefault` so users can override.
|
||||
- GPU engines need `/dev/nvidia*`. The module sets no `PrivateDevices` or `DeviceAllow`, so that access works today. Document that adding such hardening would break CUDA engines.
|
||||
- Document an example using `pkgs.whisper-cpp` and a model fetched with `pkgs.fetchurl`, or a path under `/var/lib/archivr-server/models`.
|
||||
|
||||
**Docker (`Dockerfile`, `docker-compose.yml`)**:
|
||||
- `Dockerfile`: add `ARCHIVR_FFMPEG=/usr/bin/ffmpeg` to the existing `ENV` block (ffmpeg is already apt-installed). Ship no engines; the image is CPU-only Debian bookworm.
|
||||
- `docker-compose.yml`: add commented examples for `ARCHIVR_TRANSCRIBE_ENGINES`, `ARCHIVR_PHONON2_CLI`, `ARCHIVR_WHISPER_CLI`/`ARCHIVR_WHISPER_MODEL`, and a commented read-only volume `./models:/models:ro`.
|
||||
- README: a derived-image example for Phonon-2 on CPU, using the vendor's documented commands. The image already has `python3`/`venv`:
|
||||
```dockerfile
|
||||
FROM archivr:latest
|
||||
RUN python3 -m venv /opt/transcribe && \
|
||||
/opt/transcribe/bin/pip install --no-deps torch --index-url https://download.pytorch.org/whl/cpu && \
|
||||
/opt/transcribe/bin/pip install fermion-research torch safetensors soundfile scipy zstandard
|
||||
ENV ARCHIVR_TRANSCRIBE_ENGINES=phonon2 ARCHIVR_PHONON2_CLI=/opt/transcribe/bin/fermion
|
||||
```
|
||||
Note the cache volume (`phonon-cache` in the vendor examples), so weights survive container recreation. GPU containers need the NVIDIA container toolkit and a CUDA base image, which is out of scope.
|
||||
|
||||
**Docs to update in the implementation change** (repo rule: behaviour docs move with the code):
|
||||
- `docs/README.md`:
|
||||
- a "Local transcription (optional)" subsection under Supported Inputs → YouTube subtitles: the engine table, a setup example per engine, the disk note (~115 MB of temp WAV per hour), the English-only note for Phonon-2;
|
||||
- a new `#### Local transcription` env table next to `#### LLM providers`;
|
||||
- NixOS `environment`/`environmentFile` options;
|
||||
- the Docker example.
|
||||
- `ARCHIVR-MENTAL-MODEL.md`: a node for the transcription step in the LLM-summary mermaid diagram, the `transcribed` kind and `transcription` origin, and a "Where To Edit" row for `transcriber.rs`.
|
||||
- `AGENTS.md`: add the new env vars to the "External tools by env var" bullet, `transcriber.rs`/`process.rs`/`env_config.rs` to Important Files, and the script contract under the "CLI providers parse a file" convention.
|
||||
|
||||
---
|
||||
|
||||
## 11. Tests
|
||||
|
||||
No real models, no network and no GPU in tests. Engines are faked either in-process (the `Transcriber` trait) or with executable `#!/bin/sh` stubs written into a `tempfile` dir, following the `fake_yt_dlp` pattern in `ytdlp.rs` (`std::fs::write` then `set_permissions(0o755)` under `#[cfg(unix)]`). Tests that mutate env vars must hold a module-level lock, like `ENV_LOCK` in `summarizer.rs`. Prefer the config-based constructors (`transcriber_from_config`) so most tests don't touch env at all.
|
||||
|
||||
**`process.rs`**
|
||||
- `run_with_timeout_drains_large_stderr_without_deadlock`: `sh -c 'head -c 1000000 /dev/zero | tr "\0" x >&2; echo ok'` with a 30 s timeout returns `stdout == "ok\n"`.
|
||||
- `run_with_timeout_kills_overrunning_child_and_marks_timeout`: `sleep 30`, 1 s timeout; the error chain contains `ProcessTimedOut` and it returns in under 5 s.
|
||||
- `run_with_timeout_reports_nonzero_exit_with_stderr_tail`.
|
||||
- The existing `run_cli_*` tests in `summarizer.rs` stay green unchanged.
|
||||
|
||||
**`env_config.rs`**: a `resolve_cli` priority test, if moving it leaves the existing summarizer tests without coverage. Otherwise, move the existing tests along with the helpers.
|
||||
|
||||
**`transcriber.rs`**
|
||||
- `whisper_cpp_args_with_language_hint`: exact vector `[-m, M, -f, W, -l, de, -ovtt, -oj, -of, P, -np]`. `whisper_cpp_args_without_hint_uses_auto`.
|
||||
- `whisper_language_hint_only_for_two_letter_bases`: `de-orig`→`de`, `en-GB`→`en`, `yue`→None, `zh-Hans`→`zh`, None→None, `x;rm`→None.
|
||||
- `script_args_follow_contract`: `--input`, `--output`, `--model`, plus `--language` only when a hint is given.
|
||||
- `phonon2_args_are_transcribe_model_wav_json`.
|
||||
- `ffmpeg_resample_args_are_16k_mono_pcm`: contains `-map 0:a:0`, `-ac 1`, `-ar 16000`, `-c:a pcm_s16le`, `-nostdin`; the output path is last.
|
||||
- `phonon2_refuses_known_non_english_language`: `de`, `de-orig` and `pt-BR` → false.
|
||||
- `phonon2_accepts_english_variants_and_unknown`: `en`, `en-US`, `en-orig` and `None` → true.
|
||||
- `allowlist_gating_for_whisper_and_parakeet`: `Some(["en","de"])` accepts `de-orig` and refuses `fr`; `None` accepts all.
|
||||
- `parse_language_list_trims_lowercases_dedups`; `enabled_engine_kinds_ignores_unknown_and_duplicates` (env-locked).
|
||||
- `transcriber_from_env_whisper_missing_model_names_variable`; `transcriber_from_env_whisper_script_backend_requires_cli`; `transcriber_from_env_rejects_bad_backend`; `transcriber_from_env_parakeet_missing_cli_names_variable`; `transcriber_from_env_phonon2_defaults_to_fermion_and_phonon_2`; `transcriber_from_env_rejects_unknown_kind`; `transcriber_from_env_timeout_default_and_override`. All env-locked; clear every `ARCHIVR_*TRANSCRIBE*`/engine var before and after.
|
||||
- `request_from_env_rejects_configured_but_not_enabled_engine`: the message contains `ARCHIVR_TRANSCRIBE_ENGINES`.
|
||||
- `available_transcribers_lists_enabled_and_configured_only`: whisper enabled without a model is left out; phonon2 is listed with `english_only == true`.
|
||||
- `format_vtt_timestamp_formats_and_clamps`: `3661.5` → `01:01:01.500`, `-1.0` → `00:00:00.000`.
|
||||
- `phonon_json_to_vtt_from_segments_fixture`. Use the real captured sample (§4.5); output reduced with `subtitle_to_transcript` equals the joined segment texts. `phonon_json_to_vtt_falls_back_to_single_cue_from_text`. `phonon_json_to_vtt_rejects_non_json`.
|
||||
- `select_audio_source_prefers_existing_archived_media`: an mp4 artifact whose file exists → Some. A missing file → None. An `html` primary → None.
|
||||
- `transcribe_entry_with_stub_engine_registers_transcribed_artifact`:
|
||||
- Scratch archive with a youtube/video entry and an existing `raw/…mp4` primary.
|
||||
- Stub ffmpeg writes a 44-byte header plus zeros to its last argument.
|
||||
- Stub `whisper-cli` parses `-of <prefix>` and writes `<prefix>.vtt` (`WEBVTT\n\n00:00:00.000 --> 00:00:02.000\nhello world\n`) and `<prefix>.json` (`{"result":{"language":"en"}}`).
|
||||
- Assert: one `subtitle` artifact; `metadata_json` has `kind:"transcribed"`, `origin:"transcription"`, `engine:"whisper"`, a file-name-only `model`, `language:"en"`; blob MIME `text/vtt`; `store/temp/transcribe-*` is gone.
|
||||
- `transcribe_entry_timeout_cleans_temp_and_returns_timeout_copy`: the stub engine runs `sleep 30`, `timeout_secs = 1`; `transcription_user_message` contains "timed out"; the temp dir is gone.
|
||||
- `transcribe_entry_engine_failure_message_is_sanitized`: the stub writes `/secret/path` to stderr and exits 1; the user message does not contain `/secret/path`.
|
||||
- `transcribe_entry_empty_transcript_is_no_speech`: the stub writes only `WEBVTT\n`; the result carries `NO_SUBTITLES_AFTER_TRANSCRIPTION_MESSAGE`.
|
||||
- `transcribe_entry_skips_when_usable_subtitle_already_exists`: returns 0 and never spawns the engine (point the stub path at a non-existent file; spawning it would error).
|
||||
- `transcribe_entry_without_audio_and_non_http_url_fails_with_no_audio_copy`.
|
||||
- `transcription_slot_serializes_jobs`: two threads using an in-process fake that records overlap; overlap is never observed.
|
||||
|
||||
**`downloader/ytdlp.rs`**
|
||||
- `original_language_from_metadata_prefers_language_field_then_orig_key`.
|
||||
- `audio_only_args_bestaudio_without_extract`: contains `-f bestaudio/best` and `--no-playlist`; never `-x`.
|
||||
- `subtitle_kind_transcribed_round_trips`.
|
||||
|
||||
**`subtitles.rs`**
|
||||
- `track_rank_places_transcribed_below_manual_above_auto` (and update the existing rank test).
|
||||
- `register_transcript_artifact_writes_engine_metadata_and_dedups` (run twice → one row).
|
||||
- `fetch_outcome_reports_original_language_from_existing_artifacts`: an entry with a non-HTTP canonical URL and an existing artifact whose metadata has `original_language:"de"` gives `Ok(SubtitleFetchOutcome { added: 0, original_language: Some("de") })` without yt-dlp.
|
||||
- Update `fetch_subtitles_for_entry_skips_non_youtube_and_non_http_entries` for the new return type.
|
||||
|
||||
**`summarizer.rs`** (in-process fake: `struct FakeTranscriber { vtt: &'static str, calls: AtomicUsize }` implementing `Transcriber`; it writes `out_dir/transcript.vtt`). Reuse `youtube_summary_fixture`, with the canonical URL non-HTTP so the subtitle fetch returns zero without yt-dlp, and a stub ffmpeg via `TranscriptionSettings`:
|
||||
- `subtitle_fetch_with_transcriber_uses_transcribed_track`: content starts with `Transcript (en, transcribed subtitles):`; the fake was called once.
|
||||
- `subtitle_fetch_without_transcriber_keeps_no_subtitles_error` (regression).
|
||||
- `transcriber_not_called_when_usable_subtitles_exist`.
|
||||
- `youtube_summary_digest_changes_when_transcript_added`.
|
||||
- `phonon2_non_english_original_language_fails_before_audio_work`. Seed an unusable subtitle artifact with `original_language:"de"` so the fetch outcome carries it; the fake records zero calls and the error carries the language copy.
|
||||
|
||||
**`routes.rs`**
|
||||
- `transcription_engines_endpoint_requires_user`: a guest gets 401.
|
||||
- `transcription_engines_endpoint_lists_enabled_engines`: env-locked, `ARCHIVR_TRANSCRIBE_ENGINES=phonon2`, `ARCHIVR_PHONON2_CLI=/usr/bin/false` → one item with `english_only: true`.
|
||||
- `summary_post_rejects_unconfigured_transcribe_engine_with_400`: `{"provider":"codex_cli","transcribe_engine":"whisper"}` with whisper enabled but no model. Expect 400, the body names `ARCHIVR_WHISPER_MODEL`, and no summary row exists.
|
||||
- `summary_post_rejects_engine_not_enabled`.
|
||||
- `youtube_summary_with_stub_transcription_reaches_provider`:
|
||||
- Env-locked. `make_test_youtube_entry` with the `youtube-test:offline` URL and a primary mp4 artifact whose file exists.
|
||||
- Stub ffmpeg and a stub whisper-cli (as above) through env vars; `ARCHIVR_CODEX_CLI=/usr/bin/false`.
|
||||
- POST with `transcribe_engine:"whisper"` → 202. Poll until `failed`.
|
||||
- Assert `error_text` is neither `NO_SUBTITLES_SUMMARY_MESSAGE` nor a transcription copy. That proves transcription succeeded and the provider ran. Also assert a `transcribed` subtitle artifact exists and the row's `input_sha256` is no longer the placeholder.
|
||||
- `summary_failure_error_text_prefers_transcription_copy`.
|
||||
- Existing tests (`youtube_summary_without_subtitles_fails_row_with_clear_message`, `summary_preflight_returns_safe_message_for_unsupported_video_content`) stay green unchanged.
|
||||
|
||||
**Frontend**: no ContextRail component test exists, so none is added; `bun test` must stay green. If one has been added by then, extend it: the select is hidden when the engines list is empty and for non-YouTube entries, and `transcribe_engine` is sent only when one is selected.
|
||||
|
||||
**Manual smoke test** (documented, not CI). On a host with an engine installed:
|
||||
1. Capture a YouTube video with `--no-subtitles` that has no captions, or delete its subtitle artifacts in a scratch archive.
|
||||
2. Enable the engine. Request a summary with that engine selected.
|
||||
3. Check the `subtitle` row's `metadata_json`, the VTT in `raw/`, the summary content label, and that `store/temp/` is empty.
|
||||
4. Repeat with Phonon-2 on a non-English video and expect the language copy.
|
||||
5. Verify the whisper.cpp flags and JSON `result.language` key, the Phonon `--json` keys, and the Parakeet wrapper against the installed versions.
|
||||
|
||||
---
|
||||
|
||||
## 12. Open questions
|
||||
|
||||
1. **GPU vs CPU defaults.** Should `available_transcribers` show a hardware hint, such as an "(slow on CPU)" label? That needs probing `nvidia-smi` or Metal, which is out of scope for now.
|
||||
2. **Long-video chunking.** whisper.cpp and Phonon-2 chunk internally. Parakeet wrappers must chunk themselves (Appendix A.2 notes this). Should archivr pre-split the WAV with ffmpeg (`-f segment -segment_time 600`) and stitch the cue offsets, so wrappers can stay naive? This adds complexity and is deferred.
|
||||
3. **Concurrency cap.** One job per process is fixed here. Should it be configurable (`ARCHIVR_TRANSCRIBE_CONCURRENCY`) for multi-GPU hosts? Should waiting jobs be visible ("queued") in the UI? That needs a status that does not exist today.
|
||||
4. **Persistent engine servers.** Phonon-2's `fermion serve` (OpenAI-compatible `/v1/audio/transcriptions`, 32 MB request cap, `verbose_json` with segments) and whisper.cpp's server would avoid the 10–40 s load per job. They need an HTTP client path and chunking under 32 MB (~17 min of 16 kHz mono s16 WAV). That falls under the "no cloud/HTTP ASR" non-goal for now, even when the server is localhost.
|
||||
5. **Re-transcription and engine switching.** There is no UI to drop a transcribed track or prefer another engine. This could become a "Re-transcribe with…" action that deletes the `transcription`-origin artifact (needs an artifact-delete path that keeps blob refcounts correct).
|
||||
6. **Fetching subtitles after a transcript exists.** The `fetch_subtitles_for_entry` re-check sees the transcribed track and never contacts YouTube again, even if creator captions are added later. One option is to count only non-`transcribed` artifacts in that re-check. This is deferred; it trades extra yt-dlp calls for freshness.
|
||||
7. **Capture-time transcription** and **non-YouTube audio/video** (the generic `primary_media` path): natural extensions once this path is proven.
|
||||
8. **Leftover `temp/transcribe-*` after a crash.** A startup sweep of stale `temp/` children (older than 24 h) would cover captures too. This is a separate change.
|
||||
9. **Whisper language detection with no hint.** whisper.cpp detects the language from the first 30 s; a wrong detection hurts mixed-language videos. Should archivr pass `-l en` when the title is ASCII-only? No for now; it's a heuristic.
|
||||
|
||||
---
|
||||
|
||||
## Appendix A: reference wrapper scripts (script contract §4.6)
|
||||
|
||||
These are **reference sketches and have not been run** [INFERENCE: check the APIs against the installed library versions]. Users install them anywhere and point `ARCHIVR_WHISPER_CLI` (with `ARCHIVR_WHISPER_BACKEND=script`) or `ARCHIVR_PARAKEET_CLI` at them. Archivr does not ship them.
|
||||
|
||||
Shared helpers used by all three scripts:
|
||||
|
||||
```python
|
||||
def ts(s):
|
||||
s = max(0.0, float(s)); h = int(s // 3600); m = int(s % 3600 // 60)
|
||||
return f"{h:02d}:{m:02d}:{s % 60:06.3f}"
|
||||
|
||||
def write_vtt(path, cues): # cues: iterable of (start, end, text)
|
||||
with open(path, "w", encoding="utf-8") as f:
|
||||
f.write("WEBVTT\n\n")
|
||||
for start, end, text in cues:
|
||||
text = " ".join(text.split())
|
||||
if text:
|
||||
f.write(f"{ts(start)} --> {ts(end)}\n{text}\n\n")
|
||||
```
|
||||
|
||||
### A.1 faster-whisper
|
||||
|
||||
```python
|
||||
#!/usr/bin/env python3
|
||||
import argparse
|
||||
from faster_whisper import WhisperModel
|
||||
# + ts/write_vtt from above
|
||||
|
||||
p = argparse.ArgumentParser()
|
||||
p.add_argument("--input", required=True); p.add_argument("--output", required=True)
|
||||
p.add_argument("--model", required=True); p.add_argument("--language")
|
||||
a = p.parse_args()
|
||||
model = WhisperModel(a.model, device="auto", compute_type="default")
|
||||
segments, info = model.transcribe(a.input, language=a.language, vad_filter=True)
|
||||
write_vtt(a.output, ((s.start, s.end, s.text) for s in segments))
|
||||
open(a.output + ".lang", "w").write(info.language)
|
||||
```
|
||||
|
||||
### A.2 Parakeet via NeMo
|
||||
|
||||
```python
|
||||
#!/usr/bin/env python3
|
||||
import argparse
|
||||
import nemo.collections.asr as nemo_asr
|
||||
# + ts/write_vtt from above
|
||||
|
||||
p = argparse.ArgumentParser()
|
||||
p.add_argument("--input", required=True); p.add_argument("--output", required=True)
|
||||
p.add_argument("--model", required=True); p.add_argument("--language") # ignored; v3 auto-detects
|
||||
a = p.parse_args()
|
||||
model = nemo_asr.models.ASRModel.from_pretrained(model_name=a.model)
|
||||
# Long audio: full attention has a maximum single-pass length (~24 min). For longer files either
|
||||
# switch to local attention (model.change_attention_model("rel_pos_local_attn", [256, 256])) or
|
||||
# split the WAV into chunks and offset the timestamps. [INFERENCE: verify for the chosen model]
|
||||
out = model.transcribe([a.input], timestamps=True)
|
||||
segs = out[0].timestamp["segment"]
|
||||
write_vtt(a.output, ((s["start"], s["end"], s["segment"]) for s in segs))
|
||||
```
|
||||
|
||||
### A.3 Parakeet via parakeet-mlx (Apple silicon)
|
||||
|
||||
```python
|
||||
#!/usr/bin/env python3
|
||||
import argparse
|
||||
from parakeet_mlx import from_pretrained
|
||||
# + ts/write_vtt from above
|
||||
|
||||
p = argparse.ArgumentParser()
|
||||
p.add_argument("--input", required=True); p.add_argument("--output", required=True)
|
||||
p.add_argument("--model", required=True); p.add_argument("--language")
|
||||
a = p.parse_args()
|
||||
model = from_pretrained(a.model) # e.g. mlx-community/parakeet-tdt-0.6b-v3
|
||||
result = model.transcribe(a.input)
|
||||
write_vtt(a.output, ((s.start, s.end, s.text) for s in result.sentences))
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Appendix B: file-by-file change list (implementation order)
|
||||
|
||||
1. `crates/archivr-core/src/env_config.rs` (new): move `required_env`, `env_or`, `optional_env`, `env_timeout`, `resolve_cli` from `summarizer.rs` as `pub(crate)`; update `summarizer.rs`.
|
||||
2. `crates/archivr-core/src/process.rs` (new): `run_with_timeout`, `ProcessOutput`, `ProcessTimedOut`; `summarizer::run_cli` delegates to it.
|
||||
3. `crates/archivr-core/src/downloader/ytdlp.rs`: `SubtitleKind::Transcribed`; `pub` `language_base` and `is_safe_language_code`; `original_language_from_metadata`; `download_audio_for_transcription` and `audio_only_args`.
|
||||
4. `crates/archivr-core/src/subtitles.rs`: `SUBTITLE_ORIGIN_TRANSCRIPTION`; `SubtitleFetchOutcome` and the new `fetch_subtitles_for_entry` return type; `insert_subtitle_rows` refactor; `register_transcript_artifact`; ranking table update.
|
||||
5. `crates/archivr-core/src/transcriber.rs` (new): everything in §6; `lib.rs` module declarations.
|
||||
6. `crates/archivr-core/src/summarizer.rs`: the `build_summary_input_with_subtitle_fetch` signature and body (§3.1); `NO_SUBTITLES_AFTER_TRANSCRIPTION_MESSAGE`.
|
||||
7. `crates/archivr-server/src/routes.rs`: `SummaryRequestBody.transcribe_engine`, preflight validation, background wiring, `transcription_engines_handler` plus route, `summary_failure_error_text` order.
|
||||
8. `frontend/src/api.js`, `frontend/src/components/ContextRail.jsx`, `frontend/src/styles.css`.
|
||||
9. `flake.nix`, `modules/nixos/archivr-server.nix`, `Dockerfile`, `docker-compose.yml`.
|
||||
10. `docs/README.md`, `ARCHIVR-MENTAL-MODEL.md`, `AGENTS.md`.
|
||||
11. Run `cargo build`, `cargo test`, `cd frontend && bun test`, then the manual smoke test (§11).
|
||||
|
||||
---
|
||||
|
||||
## Implementation deviations
|
||||
|
||||
The implementation follows this spec except where listed. Order of the summary path, as implemented: archived subtitles (preflight) → subtitles fetched from the original video → local transcription (only if both give nothing and an engine was requested) → error.
|
||||
|
||||
- **D1. yt-dlp audio fallback uses `ytdlp.rs`'s private `run_with_timeout(Command, Option<Duration>)`**, not `process::run_with_timeout`. yt-dlp must be built with the private `yt_dlp_command(&resolve_yt_dlp())`, which returns a `Command`. So the signature is `download_audio_for_transcription(.., timeout: Duration)`, and a timeout there is recognised by checking the job deadline after the error rather than by `ProcessTimedOut`.
|
||||
- **D2. The downloaded audio file is found with the existing `collect_staged_outputs(temp_dir, "<key>.audio", None).media`**, which already skips `.part`, `.ytdl`, `.temp` and `cookies.txt`.
|
||||
- **D3. Sentinel detection.** `ProcessTimedOut` is the *root* error with a message on top (`anyhow::Error::new(ProcessTimedOut{secs}).context("{exe} timed out after {secs}s")`); `TranscriptionUserMessage` is attached as *context*. Both detectors use `error.chain().find_map(downcast_ref).or_else(|| error.downcast_ref())`, because context layers are only reachable through `anyhow::Error::downcast_ref`. Unit tests pin this.
|
||||
- **D4. Stored `model` is reduced to its file name only if it contains `\`, is absolute, or exists on disk**, instead of "contains `/`", so Hugging Face ids such as `nvidia/parakeet-tdt-0.6b-v3` are kept. A relative path that doesn't exist from the server's working directory is stored as-is.
|
||||
- **D5. `is_safe_language_code` became `pub(crate)`, not `pub`;** `language_base` was already `pub(crate)`. Both are only used inside the crate.
|
||||
- **D6. The phonon2 truncation warning is `warn: phonon2 reported truncated segments`**, without the entry uid (the engine adapter doesn't have it). The `info: transcribed {uid} …` and `warn: transcription {uid}: …` lines name the entry.
|
||||
- **D7. Phonon JSON.** A real sample was captured (see "Verified facts"), pasted as `PHONON2_SAMPLE_JSON` in the `transcriber.rs` tests, and `phonon_json_to_vtt` is tested against it. The parser keeps the tolerant order (segments → `words` grouped into cues of ≤7 s / ≤84 chars → `text` as one cue) and accepts `text`/`word` for word text and `start`/`end` (plus `start_s`/`start_time` variants) for times.
|
||||
- **D8. ContextRail sends `transcribe_engine` only while the selector is visible** (engines non-empty and the entry is a YouTube video), so a stale session choice can't cause a 400 on other entries.
|
||||
- **D9. The dev shell adds `pkgs.ffmpeg` only;** `whisper-cpp` is left to `nix shell nixpkgs#whisper-cpp`.
|
||||
- **D10. Non-zero exit message from `process::run_with_timeout` is `"{exe} exited with {status}: …{last ≤400 chars of the stderr tail}"`.** The old `run_cli` quoted the *first* 400 chars; the tail holds the useful error.
|
||||
- **D11. Reader threads after exit.** Once the child exits, the runner waits for the stdout/stderr reader channels for at most the remaining budget, then reports a timeout, so a grandchild that keeps a pipe open can't hang the job.
|
||||
- **D12. Error copy when an engine was requested.** The user's step (4) "error" is refined per §8.1: when an engine was tried, the transcription-specific copies replace `NO_SUBTITLES_SUMMARY_MESSAGE`. The plain no-subtitles copy is still used whenever no engine was requested or the feature is off.
|
||||
- **D13. The "original language unknown" warning is logged by `transcribe_entry`** (`warn: {kind}: original language unknown for {entry_uid}; assuming it is supported`) rather than inside `supports_language`, which stays pure and has no uid.
|
||||
- **D14. Script engines (Whisper `script`, Parakeet) get the same validated two-letter hint as whisper.cpp** (`whisper_language_hint`), or no `--language` at all.
|
||||
- **D15. If moving the transcript into `raw/` fails**, the job fails with the "produced no subtitle file" copy (§8.1 has no dedicated row for it). DB errors while registering propagate without a user copy, like other DB failures.
|
||||
- **D16. `run_cli` is a thin adapter over `process::run_with_timeout`**; `summarizer.rs` no longer imports `io::Write`, `process::{Command, Stdio}`, `sync::mpsc` or `thread`.
|
||||
|
||||
### Verified facts
|
||||
|
||||
- **Phonon-2 (`fermion-research` 0.2.9, MLX backend on Apple silicon, 2026-10-05).** `fermion transcribe phonon-2 <wav> --json` prints one JSON object with the keys `text`, `model` (`"FermionResearch/Phonon-2"`), `profile`, `backend`, `engine`, `duration_seconds`, `decode_seconds`, `wall_seconds`, `segment_count`, `segments` (`[{id, start, end, text}]`), `words` (`[{text, start, end}]`) and `truncated` (bool). The first run downloaded and verified the weights into `~/.cache/fermion/speech/…`. The package metadata declares no licence, so the CLI licence is still [UNKNOWN].
|
||||
- **whisper.cpp flags and the `result.language` JSON key** are pinned by unit tests and checked during the orchestrator's live smoke run against nixpkgs `whisper-cpp` 1.8.3.
|
||||
6
flake.lock
generated
6
flake.lock
generated
|
|
@ -2,11 +2,11 @@
|
|||
"nodes": {
|
||||
"nixpkgs": {
|
||||
"locked": {
|
||||
"lastModified": 1787360063,
|
||||
"narHash": "sha256-dt4WdcvsA8/RCe+VZZwqU0X+XMM3wBbGCWA0/sFWzGo=",
|
||||
"lastModified": 1783776592,
|
||||
"narHash": "sha256-UgCQzxeWI75XM8G+hPrPh+MKzEPjG3SpAj7dtqSbksA=",
|
||||
"owner": "nixos",
|
||||
"repo": "nixpkgs",
|
||||
"rev": "2c423e03bbafcff28bfadc6781a4a8257f205cb5",
|
||||
"rev": "e7a3ca8092b61ff85b6a45bf863ea2b2d6a661b3",
|
||||
"type": "github"
|
||||
},
|
||||
"original": {
|
||||
|
|
|
|||
111
flake.nix
111
flake.nix
|
|
@ -92,89 +92,6 @@
|
|||
cp -r . $out/
|
||||
'';
|
||||
};
|
||||
# yt-dlp — pinned to a specific GitHub release rather than pulled through
|
||||
# nixpkgs. Rationale: YouTube frequently rotates player-signature/API
|
||||
# surfaces, and yt-dlp ships updates on a days-to-weeks cadence; even
|
||||
# nixos-unstable often lags by months. When the binary is stale,
|
||||
# captures fail with HTTP 403 on formats the old client can't
|
||||
# authenticate. Fetching the zipapp directly (a Python zipapp with a
|
||||
# `#!/usr/bin/env python3` shebang) lets us bump the version + hash in
|
||||
# one place without waiting on nixpkgs. Wrapped so `python3` and
|
||||
# `ffmpeg` — the two runtime deps for muxed downloads — are always on
|
||||
# PATH regardless of the caller's environment.
|
||||
#
|
||||
# Bumping: replace `version`, then run `nix hash file <url>` on the
|
||||
# new zipapp URL and paste the sri output into `hash`.
|
||||
ytDlp = pkgs.stdenv.mkDerivation {
|
||||
pname = "yt-dlp";
|
||||
version = "2026.08.19";
|
||||
src = pkgs.fetchurl {
|
||||
url = "https://github.com/yt-dlp/yt-dlp/releases/download/2026.08.19/yt-dlp";
|
||||
hash = "sha256-H6ZzPDfqb7Ucma2P54Xnt+XzJGybmAIwMp1Pty7Y1NY=";
|
||||
};
|
||||
dontUnpack = true;
|
||||
nativeBuildInputs = [ pkgs.makeWrapper ];
|
||||
installPhase = ''
|
||||
mkdir -p $out/bin
|
||||
install -m 0755 $src $out/bin/yt-dlp
|
||||
wrapProgram $out/bin/yt-dlp \
|
||||
--prefix PATH : ${lib.makeBinPath [ pkgs.python312 pkgs.ffmpeg ]}
|
||||
'';
|
||||
};
|
||||
# Frontend: per-system hash for the node_modules FOD.
|
||||
# bun installs platform-specific native binaries (esbuild, rollup),
|
||||
# so the hash differs between systems.
|
||||
# To compute the hash for a new system, set its entry to
|
||||
# "sha256-AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA=" and run:
|
||||
# nix build .#archivr-server 2>&1 | grep "got:"
|
||||
# then paste the reported hash here.
|
||||
frontendDepsHash =
|
||||
{
|
||||
"aarch64-darwin" = "sha256-QYmiCaORbrWPVaM9xXViCZChSxwObRCjlrM03zukjQ0=";
|
||||
"x86_64-linux" = "sha256-AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA=";
|
||||
"aarch64-linux" = "sha256-AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA=";
|
||||
}
|
||||
.${system} or "sha256-AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA=";
|
||||
|
||||
# FOD: fetch npm deps via bun. Network is allowed; output is hashed.
|
||||
frontendDeps = pkgs.stdenv.mkDerivation {
|
||||
pname = "archivr-frontend-deps";
|
||||
version = "0.1.0";
|
||||
src = ./frontend;
|
||||
nativeBuildInputs = [ pkgs.bun ];
|
||||
buildPhase = ''
|
||||
export HOME=$TMPDIR
|
||||
bun install --frozen-lockfile
|
||||
'';
|
||||
installPhase = ''
|
||||
cp -r node_modules $out
|
||||
'';
|
||||
outputHash = frontendDepsHash;
|
||||
outputHashAlgo = "sha256";
|
||||
outputHashMode = "recursive";
|
||||
};
|
||||
|
||||
# Build the Vite bundle using the pre-fetched node_modules.
|
||||
# Source files in frontend/src/ are jj-tracked and flow through automatically;
|
||||
# only the deps hash (above) needs updating when bun.lock/package.json changes.
|
||||
frontendStatic = pkgs.stdenv.mkDerivation {
|
||||
pname = "archivr-frontend-static";
|
||||
version = "0.1.0";
|
||||
src = ./frontend;
|
||||
nativeBuildInputs = [ pkgs.nodejs ];
|
||||
buildPhase = ''
|
||||
export HOME=$TMPDIR
|
||||
export BABEL_CACHE_PATH=$TMPDIR/babel-cache
|
||||
cp -r ${frontendDeps} node_modules
|
||||
chmod -R u+w node_modules
|
||||
node node_modules/vite/bin/vite.js build --outDir dist
|
||||
'';
|
||||
installPhase = ''
|
||||
cp -r dist $out
|
||||
'';
|
||||
dontFixup = true;
|
||||
};
|
||||
|
||||
version = "0.1.0";
|
||||
src = pkgs.lib.cleanSource ./.;
|
||||
cargoLock = {
|
||||
|
|
@ -222,10 +139,9 @@
|
|||
version = "0.1.0";
|
||||
nativeBuildInputs = [ pkgs.makeWrapper ];
|
||||
buildInputs = [
|
||||
ytDlp
|
||||
pkgs.yt-dlp
|
||||
pkgs.single-file-cli
|
||||
tweetPython
|
||||
pkgs.deno
|
||||
] ++ lib.optionals pkgs.stdenv.isLinux [ pkgs.chromium ];
|
||||
phases = [ "installPhase" ];
|
||||
installPhase = ''
|
||||
|
|
@ -233,12 +149,8 @@
|
|||
cp ${archivr_cli_unwrapped}/bin/archivr $out/libexec/archivr/archivr
|
||||
cp ${./vendor/twitter/scrape_user_tweet_contents.py} $out/libexec/archivr/scrape_user_tweet_contents.py
|
||||
chmod +x $out/libexec/archivr/scrape_user_tweet_contents.py
|
||||
# Pinned Deno is the fallback JS runtime for yt-dlp's YouTube challenge
|
||||
# solver; a newer state-dir copy from `archivr yt-dlp update` takes precedence.
|
||||
makeWrapper $out/libexec/archivr/archivr $out/bin/archivr \
|
||||
--set ARCHIVR_YT_DLP ${ytDlp}/bin/yt-dlp \
|
||||
--set ARCHIVR_DENO ${pkgs.deno}/bin/deno \
|
||||
--set ARCHIVR_FFMPEG ${pkgs.ffmpeg}/bin/ffmpeg \
|
||||
--set ARCHIVR_YT_DLP ${pkgs.yt-dlp}/bin/yt-dlp \
|
||||
--set ARCHIVR_SINGLE_FILE ${pkgs.single-file-cli}/bin/single-file \
|
||||
${lib.optionalString pkgs.stdenv.isLinux "--set ARCHIVR_CHROME ${pkgs.chromium}/bin/chromium"} \
|
||||
--set ARCHIVR_TWEET_PYTHON ${tweetPython}/bin/python3 \
|
||||
|
|
@ -247,10 +159,9 @@
|
|||
--set ARCHIVR_COOKIE_EXT ${isdcac} \
|
||||
--prefix PATH : ${
|
||||
lib.makeBinPath ([
|
||||
ytDlp
|
||||
pkgs.yt-dlp
|
||||
pkgs.single-file-cli
|
||||
tweetPython
|
||||
pkgs.deno
|
||||
] ++ lib.optionals pkgs.stdenv.isLinux [ pkgs.chromium ])
|
||||
}
|
||||
'';
|
||||
|
|
@ -259,28 +170,22 @@
|
|||
pname = "archivr-server-wrapped";
|
||||
inherit version;
|
||||
nativeBuildInputs = [ pkgs.makeWrapper ];
|
||||
buildInputs = [ ytDlp tweetPython pkgs.single-file-cli pkgs.deno ] ++ lib.optionals pkgs.stdenv.isLinux [ pkgs.chromium ];
|
||||
buildInputs = [ tweetPython pkgs.single-file-cli ] ++ lib.optionals pkgs.stdenv.isLinux [ pkgs.chromium ];
|
||||
phases = [ "installPhase" ];
|
||||
installPhase = ''
|
||||
mkdir -p $out/bin $out/libexec/archivr-server $out/share/archivr-server/static
|
||||
cp ${archivr_server_unwrapped}/bin/archivr-server $out/libexec/archivr-server/archivr-server
|
||||
cp ${./vendor/twitter/scrape_user_tweet_contents.py} $out/libexec/archivr-server/scrape_user_tweet_contents.py
|
||||
chmod +x $out/libexec/archivr-server/scrape_user_tweet_contents.py
|
||||
cp -r ${frontendStatic}/* $out/share/archivr-server/static/
|
||||
# Pinned Deno is the fallback JS runtime for yt-dlp's YouTube challenge
|
||||
# solver; a newer state-dir copy from `archivr yt-dlp update` takes precedence.
|
||||
cp -r ${./crates/archivr-server/static}/* $out/share/archivr-server/static/
|
||||
makeWrapper $out/libexec/archivr-server/archivr-server $out/bin/archivr-server \
|
||||
--set ARCHIVR_STATIC_DIR $out/share/archivr-server/static \
|
||||
--set ARCHIVR_YT_DLP ${ytDlp}/bin/yt-dlp \
|
||||
--set ARCHIVR_DENO ${pkgs.deno}/bin/deno \
|
||||
--set ARCHIVR_FFMPEG ${pkgs.ffmpeg}/bin/ffmpeg \
|
||||
--set ARCHIVR_SINGLE_FILE ${pkgs.single-file-cli}/bin/single-file \
|
||||
${lib.optionalString pkgs.stdenv.isLinux "--set ARCHIVR_CHROME ${pkgs.chromium}/bin/chromium"} \
|
||||
--set ARCHIVR_TWEET_PYTHON ${tweetPython}/bin/python3 \
|
||||
--set ARCHIVR_TWEET_SCRAPER $out/libexec/archivr-server/scrape_user_tweet_contents.py \
|
||||
--set ARCHIVR_UBLOCK_EXT ${ublockLite} \
|
||||
--set ARCHIVR_COOKIE_EXT ${isdcac} \
|
||||
--prefix PATH : ${lib.makeBinPath ([ ytDlp pkgs.single-file-cli tweetPython pkgs.deno ] ++ lib.optionals pkgs.stdenv.isLinux [ pkgs.chromium ])}
|
||||
--set ARCHIVR_COOKIE_EXT ${isdcac}
|
||||
'';
|
||||
};
|
||||
archivr-all = pkgs.symlinkJoin {
|
||||
|
|
@ -349,13 +254,11 @@
|
|||
pkgs.yt-dlp
|
||||
pkgs.nushell
|
||||
pkgs.uv
|
||||
pkgs.deno
|
||||
pkgs.ffmpeg
|
||||
tweetPython
|
||||
];
|
||||
shellHook = ''
|
||||
export SHELL=${pkgs.nushell}/bin/nu
|
||||
echo "nushell dev shell active – yt-dlp, deno, uv, and tweet scraper Python on PATH"
|
||||
echo "nushell dev shell active – yt-dlp, uv, and tweet scraper Python on PATH"
|
||||
nu
|
||||
'';
|
||||
};
|
||||
|
|
|
|||
13
frontend/.storybook/main.js
Normal file
13
frontend/.storybook/main.js
Normal file
|
|
@ -0,0 +1,13 @@
|
|||
/** @type { import('@storybook/react-vite').StorybookConfig } */
|
||||
const config = {
|
||||
stories: ['../src/**/*.stories.{js,jsx,ts,tsx}'],
|
||||
addons: [
|
||||
'@storybook/addon-essentials',
|
||||
'@storybook/addon-interactions',
|
||||
],
|
||||
framework: {
|
||||
name: '@storybook/react-vite',
|
||||
options: {},
|
||||
},
|
||||
};
|
||||
export default config;
|
||||
5
frontend/.storybook/preview.js
Normal file
5
frontend/.storybook/preview.js
Normal file
|
|
@ -0,0 +1,5 @@
|
|||
import '../src/styles.css';
|
||||
|
||||
export const parameters = {
|
||||
layout: 'fullscreen',
|
||||
};
|
||||
1884
frontend/bun.lock
1884
frontend/bun.lock
File diff suppressed because it is too large
Load diff
|
|
@ -6,14 +6,24 @@
|
|||
"scripts": {
|
||||
"dev": "vite",
|
||||
"build": "vite build",
|
||||
"preview": "vite preview"
|
||||
"preview": "vite preview",
|
||||
"storybook": "storybook dev -p 6006",
|
||||
"storybook:build": "storybook build"
|
||||
},
|
||||
"dependencies": {
|
||||
"react": "^18.3.1",
|
||||
"react-dom": "^18.3.1"
|
||||
},
|
||||
"devDependencies": {
|
||||
"@storybook/addon-essentials": "^7.6.20",
|
||||
"@storybook/addon-interactions": "^7.6.20",
|
||||
"@storybook/blocks": "^7.6.20",
|
||||
"@storybook/preview-api": "^7.6.20",
|
||||
"@storybook/react": "^7.6.20",
|
||||
"@storybook/react-vite": "^7.6.20",
|
||||
"@storybook/test": "^7.6.20",
|
||||
"@vitejs/plugin-react": "^4.3.4",
|
||||
"storybook": "^7.6.20",
|
||||
"vite": "^5.4.11"
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -123,9 +123,6 @@ export default function App() {
|
|||
const [collections, setCollections] = useState([])
|
||||
const [entries, setEntries] = useState([])
|
||||
const [deletedUids, setDeletedUids] = useState(() => new Set())
|
||||
// Renames made in this session, keyed by entry_uid. Child rows live in each
|
||||
// EntryRow's private state (not `entries`), so they read titles through this.
|
||||
const [renamedTitles, setRenamedTitles] = useState(() => new Map())
|
||||
const [selectedEntryUid, setSelectedEntryUid] = useState(() => parseLocation().entry)
|
||||
const [selectedEntry, setSelectedEntry] = useState(null)
|
||||
const [selectedUids, setSelectedUids] = useState(() => {
|
||||
|
|
@ -430,9 +427,6 @@ export default function App() {
|
|||
}, [tagFilter]);
|
||||
|
||||
const handleEntryTitleChange = useCallback((entryUid, newTitle) => {
|
||||
setRenamedTitles(prev => new Map(prev).set(entryUid, newTitle))
|
||||
const cached = entryCacheRef.current.get(entryUid)
|
||||
if (cached) entryCacheRef.current.set(entryUid, { ...cached, title: newTitle })
|
||||
setEntries(prev => prev.map(e =>
|
||||
e.entry_uid === entryUid ? { ...e, title: newTitle } : e
|
||||
))
|
||||
|
|
@ -643,10 +637,6 @@ export default function App() {
|
|||
setToasts(prev => prev.filter(t => t.id !== id))
|
||||
}, [])
|
||||
|
||||
const handleChildReorderError = useCallback((message) => {
|
||||
handleToast(message, null, 'error', 'Reorder failed')
|
||||
}, [handleToast])
|
||||
|
||||
const handleIgnoreUblock = useCallback(() => {
|
||||
sessionStorage.setItem('ublockWarningIgnored', 'true')
|
||||
setUblockWarningIgnored(true)
|
||||
|
|
@ -762,10 +752,7 @@ export default function App() {
|
|||
archiveId={archiveId}
|
||||
pendingCaptures={pendingCaptures}
|
||||
deletedUids={deletedUids}
|
||||
renamedTitles={renamedTitles}
|
||||
isPublicSession={!currentUser}
|
||||
canReorder={!!currentUser?.can_reorder_children}
|
||||
onChildReorderError={handleChildReorderError}
|
||||
/>
|
||||
)}
|
||||
{view === 'runs' && <RunsView runs={runs} />}
|
||||
|
|
|
|||
|
|
@ -1,5 +1,5 @@
|
|||
async function getJson(url, options) {
|
||||
const response = await fetch(url, options);
|
||||
async function getJson(url) {
|
||||
const response = await fetch(url);
|
||||
if (!response.ok) {
|
||||
throw new Error(`${response.status} ${response.statusText}`);
|
||||
}
|
||||
|
|
@ -29,87 +29,10 @@ export async function fetchEntryDetail(archiveId, entryUid) {
|
|||
return getJson(`/api/archives/${archiveId}/entries/${entryUid}`);
|
||||
}
|
||||
|
||||
// ── Entry summaries ────────────────────────────────────────────────────────
|
||||
// Summaries are generated on demand, never at capture time. GET is safe for
|
||||
// public sessions (the server applies the same visibility gate as entry detail).
|
||||
|
||||
export async function fetchEntrySummary(archiveId, entryUid, { signal } = {}) {
|
||||
return getJson(`/api/archives/${archiveId}/entries/${entryUid}/summary`, { signal });
|
||||
}
|
||||
|
||||
// Local transcription engines that are enabled and configured on this server
|
||||
// ([{ kind, label, english_only, languages }]); an empty list means the
|
||||
// feature is off. Logged-in users only; callers treat errors as [].
|
||||
export async function fetchTranscriptionEngines({ signal } = {}) {
|
||||
return getJson('/api/summary/transcription-engines', { signal });
|
||||
}
|
||||
|
||||
// Kicks off generation. Resolves to either an existing completed summary (200)
|
||||
// or a freshly claimed pending row (202) — both carry a summary_uid, so the
|
||||
// caller polls fetchEntrySummary either way.
|
||||
// The server returns 400 with the exact missing env var name when a provider is
|
||||
// unconfigured, so its body is surfaced verbatim rather than replaced.
|
||||
// `transcribeEngine` (a kind from fetchTranscriptionEngines) is sent only when
|
||||
// non-empty; the server uses it only for YouTube videos without subtitles.
|
||||
export async function requestEntrySummary(archiveId, entryUid, { provider, force = false, includeImages = false, transcribeEngine, signal } = {}) {
|
||||
const payload = { provider, force, include_images: includeImages };
|
||||
if (typeof transcribeEngine === 'string' && transcribeEngine.trim()) {
|
||||
payload.transcribe_engine = transcribeEngine.trim();
|
||||
}
|
||||
const resp = await fetch(
|
||||
`/api/archives/${archiveId}/entries/${entryUid}/summary`,
|
||||
{
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify(payload),
|
||||
signal,
|
||||
}
|
||||
);
|
||||
if (!resp.ok) {
|
||||
// ApiError renders as { "error": "..." }; that message is the useful part
|
||||
// (e.g. "missing required environment variable: ARCHIVR_ANTHROPIC_API_KEY"),
|
||||
// so surface it verbatim instead of a generic status string.
|
||||
const detail = await resp.text();
|
||||
let message = detail.trim();
|
||||
try { message = JSON.parse(detail).error || message } catch { /* non-JSON body */ }
|
||||
throw new Error(message || `Summary request failed (${resp.status})`);
|
||||
}
|
||||
return resp.json();
|
||||
}
|
||||
|
||||
// Text artifacts are served by the same entry-artifact endpoint as previews.
|
||||
// Keep credentials explicit because this helper is also used by public/private
|
||||
// archive views, and preserve the previous concise HTTP error contract.
|
||||
export async function fetchArtifactText(src, { signal } = {}) {
|
||||
const response = await fetch(src, { credentials: 'same-origin', signal });
|
||||
if (!response.ok) throw new Error(`HTTP ${response.status}`);
|
||||
return response.text();
|
||||
}
|
||||
|
||||
export async function fetchEntryChildren(archiveId, entryUid) {
|
||||
return getJson(`/api/archives/${archiveId}/entries/${entryUid}/children`);
|
||||
}
|
||||
|
||||
// Persists a new sibling order for a parent's direct children. `childUids`
|
||||
// must be exactly the parent's current children; the server answers 400
|
||||
// when the set is stale (e.g. a sync added a video) — err.status carries it.
|
||||
export async function reorderEntryChildren(archiveId, parentEntryUid, childUids) {
|
||||
const res = await fetch(
|
||||
`/api/archives/${archiveId}/entries/${parentEntryUid}/children/order`,
|
||||
{
|
||||
method: 'PUT',
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
body: JSON.stringify({ child_uids: childUids }),
|
||||
}
|
||||
);
|
||||
if (!res.ok) {
|
||||
const body = await res.json().catch(() => ({}));
|
||||
const err = new Error(body.error || `HTTP ${res.status}`);
|
||||
err.status = res.status;
|
||||
throw err;
|
||||
}
|
||||
}
|
||||
|
||||
// Fetch multiple artifact JSON payloads for an entry in parallel.
|
||||
// Returns a Promise<Array> preserving index order.
|
||||
export function fetchEntryArtifacts(archiveId, entryUid, indices) {
|
||||
|
|
@ -143,28 +66,6 @@ export async function updateEntryTitle(archiveId, entryUid, title) {
|
|||
if (!res.ok) throw new Error(await res.text());
|
||||
}
|
||||
|
||||
// Names an X thread with a cheap model on the server and saves it as the title.
|
||||
// Resolves to { entry_uid, title }; the server's { error } text is surfaced verbatim
|
||||
// (e.g. a missing API-key variable or the provider's failure).
|
||||
export async function generateThreadTitle(archiveId, entryUid, { provider }) {
|
||||
const res = await fetch(`/api/archives/${archiveId}/entries/${entryUid}/thread-title`, {
|
||||
method: 'POST',
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
body: JSON.stringify({ provider }),
|
||||
});
|
||||
if (!res.ok) {
|
||||
const detail = await res.text();
|
||||
let message = detail.trim();
|
||||
try {
|
||||
message = JSON.parse(detail).error || message;
|
||||
} catch {
|
||||
// non-JSON body
|
||||
}
|
||||
throw new Error(message || `Title generation failed (${res.status})`);
|
||||
}
|
||||
return res.json();
|
||||
}
|
||||
|
||||
export async function fetchEntryTags(archiveId, entryUid) {
|
||||
return getJson(`/api/archives/${archiveId}/entries/${entryUid}/tags`);
|
||||
}
|
||||
|
|
@ -249,14 +150,13 @@ export async function fetchTags(archiveId) {
|
|||
export async function submitCapture(archiveId, locator, quality = null, extensions = null) {
|
||||
const payload = { locator }
|
||||
if (quality && quality !== 'best') payload.quality = quality
|
||||
// extensions: { ublock_enabled?: bool, reader_mode?: bool, cookie_ext_enabled?: bool, modal_closer_enabled?: bool, via_freedium?: bool, download_subtitles?: bool }
|
||||
// extensions: { ublock_enabled?: bool, reader_mode?: bool, cookie_ext_enabled?: bool, modal_closer_enabled?: bool, via_freedium?: bool }
|
||||
if (extensions) {
|
||||
if (typeof extensions.ublock_enabled === 'boolean') payload.ublock_enabled = extensions.ublock_enabled
|
||||
if (typeof extensions.reader_mode === 'boolean') payload.reader_mode = extensions.reader_mode
|
||||
if (typeof extensions.cookie_ext_enabled === 'boolean') payload.cookie_ext_enabled = extensions.cookie_ext_enabled
|
||||
if (typeof extensions.modal_closer_enabled === 'boolean') payload.modal_closer_enabled = extensions.modal_closer_enabled
|
||||
if (typeof extensions.via_freedium === 'boolean') payload.via_freedium = extensions.via_freedium
|
||||
if (typeof extensions.download_subtitles === 'boolean') payload.download_subtitles = extensions.download_subtitles
|
||||
if (extensions.per_item_quality && typeof extensions.per_item_quality === 'object' && Object.keys(extensions.per_item_quality).length > 0) payload.per_item_quality = extensions.per_item_quality
|
||||
if (extensions.sync === true) payload.sync = true
|
||||
}
|
||||
|
|
@ -274,24 +174,6 @@ export async function submitCapture(archiveId, locator, quality = null, extensio
|
|||
return res.json(); // { job_uid, status: "pending" }
|
||||
}
|
||||
|
||||
export async function submitTextCapture(archiveId, {title, body, mime = 'text/markdown'}) {
|
||||
const payload = { title, body };
|
||||
if (mime && mime !== 'text/markdown') payload.mime = mime;
|
||||
|
||||
const res = await fetch(`/api/archives/${archiveId}/captures/text`, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify(payload),
|
||||
});
|
||||
if (!res.ok) {
|
||||
const body = await res.json().catch(() => ({}));
|
||||
const err = new Error(body.error || `HTTP ${res.status}`);
|
||||
err.status = res.status;
|
||||
throw err;
|
||||
}
|
||||
return res.json(); // { job_uid, status: "pending" }
|
||||
}
|
||||
|
||||
// Returns { has_video: bool, qualities: string[] } e.g. { has_video: true, qualities: ["1080p","720p","480p"] }
|
||||
// Throws on network error; returns { has_video: false, qualities: [] } on non-video locators.
|
||||
export async function probeCapture(archiveId, locator) {
|
||||
|
|
@ -415,24 +297,7 @@ export async function updateInstanceSettings(patch) {
|
|||
headers: { 'Content-Type': 'application/json' },
|
||||
body: JSON.stringify(patch),
|
||||
});
|
||||
if (!res.ok) { const b = await res.json().catch(() => ({})); throw new Error(b.error || `HTTP ${res.status}`); }
|
||||
}
|
||||
|
||||
export async function getYtDlpStatus({ signal } = {}) {
|
||||
return getJson('/api/admin/yt-dlp', { signal });
|
||||
}
|
||||
|
||||
// Runs the yt-dlp + Deno update server-side (same as `archivr yt-dlp update`); can take minutes.
|
||||
// err.status carries the HTTP status (409 = another update is running).
|
||||
export async function updateYtDlp() {
|
||||
const res = await fetch('/api/admin/yt-dlp/update', { method: 'POST' });
|
||||
if (!res.ok) {
|
||||
const b = await res.json().catch(() => ({}));
|
||||
const err = new Error(b.error || `HTTP ${res.status}`);
|
||||
err.status = res.status;
|
||||
throw err;
|
||||
}
|
||||
return res.json();
|
||||
if (!res.ok) throw new Error(await res.text());
|
||||
}
|
||||
|
||||
// ── Admin helpers ─────────────────────────────────────────────────────────────
|
||||
|
|
|
|||
|
|
@ -1,5 +1,5 @@
|
|||
import { useRef, useEffect, useState, useCallback } from 'react'
|
||||
import { submitCapture, submitTextCapture, pollCaptureJob, probeCapture, probePlaylist, getInstanceSettings, uploadFile, deleteUpload } from '../api'
|
||||
import { submitCapture, pollCaptureJob, probeCapture, probePlaylist, getInstanceSettings, uploadFile, deleteUpload } from '../api'
|
||||
|
||||
let nextItemId = 1
|
||||
|
||||
|
|
@ -157,30 +157,6 @@ function makeFileItem(filename) {
|
|||
}
|
||||
}
|
||||
|
||||
function makeTextItem() {
|
||||
return {
|
||||
id: nextItemId++,
|
||||
kind: 'text',
|
||||
title: '',
|
||||
body: '',
|
||||
mime: 'text/markdown',
|
||||
// Fields present for submission-logic compatibility
|
||||
locator: '',
|
||||
quality: 'best',
|
||||
probeState: 'idle',
|
||||
probeQualities: null,
|
||||
probeHasAudio: false,
|
||||
playlistProbeState: 'idle',
|
||||
playlistInfo: null,
|
||||
playlistItems: null,
|
||||
playlistQuality: null,
|
||||
playlistExpanded: false,
|
||||
syncEnabled: false,
|
||||
error: null,
|
||||
status: 'idle',
|
||||
}
|
||||
}
|
||||
|
||||
function applyPlaylistQuality(newQ, currentItems) {
|
||||
if (newQ === 'best') {
|
||||
return currentItems.map(item => ({ ...item, quality: 'best' }))
|
||||
|
|
@ -279,7 +255,6 @@ export default function CaptureDialog({ open, archiveId, onClose, onCaptured, on
|
|||
const [cookieExtEnabled, setCookieExtEnabled] = useState(true)
|
||||
const [modalCloserEnabled, setModalCloserEnabled] = useState(true)
|
||||
const [freediumEnabled, setFreediumEnabled] = useState(true)
|
||||
const [downloadSubtitles, setDownloadSubtitles] = useState(true)
|
||||
|
||||
// Load global settings from server once on mount
|
||||
useEffect(() => {
|
||||
|
|
@ -455,30 +430,9 @@ export default function CaptureDialog({ open, archiveId, onClose, onCaptured, on
|
|||
onToastRef.current(text, null, type, headline)
|
||||
}
|
||||
|
||||
async function submitBgJob(submission, batchId) {
|
||||
async function submitBgJob(locator, quality, batchId, extraExtensions = {}) {
|
||||
const aid = archiveIdRef.current
|
||||
const id = crypto.randomUUID?.() ?? `job-${Date.now()}-${Math.random()}`
|
||||
|
||||
// Text submission
|
||||
if (submission.type === 'text') {
|
||||
try {
|
||||
const job = await submitTextCapture(aid, { title: submission.title, body: submission.body, mime: submission.mime })
|
||||
const locator = `text:${submission.title}`
|
||||
// Notify App to add skeleton + persist
|
||||
onJobStartedRef.current?.({ id, jobUid: job.job_uid, locator, archiveId: aid })
|
||||
startPolling(id, job.job_uid, locator, aid, batchId)
|
||||
} catch (e) {
|
||||
const msg = e.message || 'Submission failed.'
|
||||
onToastRef.current(msg, `text:${submission.title}`)
|
||||
settleBatch(batchId, 'failed', `text:${submission.title}`)
|
||||
}
|
||||
return
|
||||
}
|
||||
|
||||
// URL/file submission
|
||||
const locator = submission.locator
|
||||
const quality = submission.quality
|
||||
const extraExtensions = submission.extraExtensions || {}
|
||||
// Capture session options at call time (synchronous — before first await)
|
||||
const extensions = {
|
||||
ublock_enabled: ublockEnabled,
|
||||
|
|
@ -486,7 +440,6 @@ export default function CaptureDialog({ open, archiveId, onClose, onCaptured, on
|
|||
cookie_ext_enabled: cookieExtEnabled,
|
||||
modal_closer_enabled: modalCloserEnabled,
|
||||
via_freedium: freediumEnabled,
|
||||
download_subtitles: downloadSubtitles,
|
||||
...extraExtensions,
|
||||
}
|
||||
try {
|
||||
|
|
@ -513,18 +466,16 @@ export default function CaptureDialog({ open, archiveId, onClose, onCaptured, on
|
|||
// Guard against the Enter-key shortcut in CaptureRow bypassing the
|
||||
// disabled button — uploads must be complete before archiving starts.
|
||||
if (items.some(it => it.kind === 'file' && it.uploadStatus === 'uploading')) return
|
||||
const toSubmit = items.filter(it => {
|
||||
if (it.kind === 'file') return it.uploadStatus === 'done' && it.uploadLocator
|
||||
if (it.kind === 'text') return it.title.trim() && it.body.trim()
|
||||
return it.locator.trim()
|
||||
})
|
||||
const toSubmit = items.filter(it =>
|
||||
it.kind === 'file' ? (it.uploadStatus === 'done' && it.uploadLocator) : it.locator.trim()
|
||||
)
|
||||
if (toSubmit.length === 0) return
|
||||
if (toSubmit.some(it => it.kind !== 'file' && it.kind !== 'text' && hasConflict(it))) return
|
||||
if (toSubmit.some(it => it.kind !== 'file' && it.kind !== 'text' && (
|
||||
if (toSubmit.some(it => it.kind !== 'file' && hasConflict(it))) return
|
||||
if (toSubmit.some(it => it.kind !== 'file' && (
|
||||
it.probeState === 'probing' ||
|
||||
(isPlaylistSource(it.locator) && it.playlistProbeState !== 'done'))))
|
||||
return
|
||||
if (toSubmit.some(it => it.kind !== 'file' && it.kind !== 'text' && Array.isArray(it.playlistItems) && it.playlistItems.length === 0)) return
|
||||
if (toSubmit.some(it => it.kind !== 'file' && Array.isArray(it.playlistItems) && it.playlistItems.length === 0)) return
|
||||
const batchId = toSubmit.length > 1
|
||||
? (crypto.randomUUID?.() ?? `batch-${Date.now()}`)
|
||||
: null
|
||||
|
|
@ -534,13 +485,9 @@ export default function CaptureDialog({ open, archiveId, onClose, onCaptured, on
|
|||
// Capture all submission data before any state changes
|
||||
const submissions = toSubmit.map(it => {
|
||||
if (it.kind === 'file') {
|
||||
return { type: 'file', locator: it.uploadLocator, quality: 'best', extraExtensions: {} }
|
||||
}
|
||||
if (it.kind === 'text') {
|
||||
return { type: 'text', title: it.title.trim(), body: it.body, mime: it.mime }
|
||||
return { locator: it.uploadLocator, quality: 'best', extraExtensions: {} }
|
||||
}
|
||||
return {
|
||||
type: 'url',
|
||||
locator: it.locator.trim(),
|
||||
quality: it.playlistItems !== null ? null : (it.quality || 'best'),
|
||||
extraExtensions: it.playlistItems !== null
|
||||
|
|
@ -555,8 +502,8 @@ export default function CaptureDialog({ open, archiveId, onClose, onCaptured, on
|
|||
setItems([makeItem()])
|
||||
dialogRef.current?.close()
|
||||
// Submit each in background
|
||||
submissions.forEach(submission =>
|
||||
submitBgJob(submission, batchId)
|
||||
submissions.forEach(({ locator, quality, extraExtensions }) =>
|
||||
submitBgJob(locator, quality, batchId, extraExtensions)
|
||||
)
|
||||
}
|
||||
|
||||
|
|
@ -683,10 +630,8 @@ export default function CaptureDialog({ open, archiveId, onClose, onCaptured, on
|
|||
files.forEach(file => {
|
||||
const newItem = makeFileItem(file.name)
|
||||
setItems(prev => {
|
||||
// Only a normal URL row can be replaced. Text drafts deliberately use
|
||||
// an empty compatibility locator, but their title/body must survive a
|
||||
// file attachment and remain independently archivable.
|
||||
if (prev.length === 1 && !prev[0].kind && !prev[0].locator.trim()) {
|
||||
// Replace a sole empty URL row with the file item; otherwise append
|
||||
if (prev.length === 1 && prev[0].kind !== 'file' && !prev[0].locator.trim()) {
|
||||
return [newItem]
|
||||
}
|
||||
return [...prev, newItem]
|
||||
|
|
@ -742,18 +687,16 @@ export default function CaptureDialog({ open, archiveId, onClose, onCaptured, on
|
|||
|
||||
|
||||
const anyUploading = items.some(it => it.kind === 'file' && it.uploadStatus === 'uploading')
|
||||
const pendingCount = items.filter(it => {
|
||||
if (it.kind === 'file') return it.uploadStatus === 'done' && it.uploadLocator
|
||||
if (it.kind === 'text') return it.title.trim() && it.body.trim()
|
||||
return it.locator.trim()
|
||||
}).length
|
||||
const anyConflict = items.some(it => it.kind !== 'file' && it.kind !== 'text' && hasConflict(it))
|
||||
const pendingCount = items.filter(it =>
|
||||
it.kind === 'file' ? (it.uploadStatus === 'done' && it.uploadLocator) : it.locator.trim()
|
||||
).length
|
||||
const anyConflict = items.some(it => it.kind !== 'file' && hasConflict(it))
|
||||
// True if any playlist row has had all its videos deleted — archive would be a no-op.
|
||||
const anyEmptyPlaylist = items.some(it =>
|
||||
it.kind !== 'file' && it.kind !== 'text' && Array.isArray(it.playlistItems) && it.playlistItems.length === 0
|
||||
it.kind !== 'file' && Array.isArray(it.playlistItems) && it.playlistItems.length === 0
|
||||
)
|
||||
const anyProbing = items.some(it =>
|
||||
it.kind !== 'file' && it.kind !== 'text' && (
|
||||
it.kind !== 'file' && (
|
||||
it.probeState === 'probing' ||
|
||||
// For playlist sources block unless probe completed successfully:
|
||||
// idle = debounce not yet fired; probing = in flight; error = no quality data.
|
||||
|
|
@ -792,17 +735,6 @@ export default function CaptureDialog({ open, archiveId, onClose, onCaptured, on
|
|||
item={item}
|
||||
onRemove={() => removeRow(item.id)}
|
||||
/>
|
||||
) : item.kind === 'text' ? (
|
||||
<CaptureTextRow
|
||||
key={item.id}
|
||||
item={item}
|
||||
autoFocus={idx === items.length - 1}
|
||||
onTitleChange={val => setItems(prev => prev.map(it => it.id === item.id ? { ...it, title: val } : it))}
|
||||
onBodyChange={val => setItems(prev => prev.map(it => it.id === item.id ? { ...it, body: val } : it))}
|
||||
onMimeChange={val => setItems(prev => prev.map(it => it.id === item.id ? { ...it, mime: val } : it))}
|
||||
onRemove={() => removeRow(item.id)}
|
||||
onSubmit={handleArchive}
|
||||
/>
|
||||
) : (
|
||||
<CaptureRow
|
||||
key={item.id}
|
||||
|
|
@ -849,12 +781,6 @@ export default function CaptureDialog({ open, archiveId, onClose, onCaptured, on
|
|||
</svg>
|
||||
Upload file
|
||||
</button>
|
||||
<button type="button" className="capture-add-row capture-add-text" onClick={() => setItems(prev => [...prev, makeTextItem()])}>
|
||||
<svg viewBox="0 0 16 16" fill="none" stroke="currentColor" strokeWidth="1.75" strokeLinecap="round" strokeLinejoin="round">
|
||||
<path d="M2 3h12M2 7h12M2 11h8"/>
|
||||
</svg>
|
||||
Add text
|
||||
</button>
|
||||
</div>
|
||||
|
||||
{/* ── Advanced options ────────────────────────────── */}
|
||||
|
|
@ -959,22 +885,6 @@ export default function CaptureDialog({ open, archiveId, onClose, onCaptured, on
|
|||
<span className="ext-toggle-knob" />
|
||||
</button>
|
||||
</label>
|
||||
<label className="capture-ext-row" style={{ marginTop: 8 }}>
|
||||
<span className="capture-ext-label">
|
||||
<span className="capture-ext-name">Download subtitles</span>
|
||||
<span className="capture-ext-desc">Save YouTube subtitles (manual preferred, auto-generated fallback) so videos can be summarized</span>
|
||||
</span>
|
||||
<button
|
||||
type="button"
|
||||
role="switch"
|
||||
aria-checked={downloadSubtitles}
|
||||
className={`ext-toggle ext-toggle--sm${downloadSubtitles ? ' ext-toggle--on' : ''}`}
|
||||
onClick={() => setDownloadSubtitles(v => !v)}
|
||||
aria-label="Toggle subtitle download for this capture"
|
||||
>
|
||||
<span className="ext-toggle-knob" />
|
||||
</button>
|
||||
</label>
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
|
|
@ -1229,63 +1139,3 @@ function CaptureFileRow({ item, onRemove }) {
|
|||
</div>
|
||||
)
|
||||
}
|
||||
|
||||
function CaptureTextRow({ item, autoFocus, onTitleChange, onBodyChange, onMimeChange, onRemove, onSubmit }) {
|
||||
const titleInputRef = useRef(null)
|
||||
|
||||
useEffect(() => {
|
||||
if (autoFocus) {
|
||||
titleInputRef.current?.focus()
|
||||
}
|
||||
}, [autoFocus]) // eslint-disable-line react-hooks/exhaustive-deps
|
||||
|
||||
return (
|
||||
<div className="capture-row capture-text-row">
|
||||
<div className="capture-row-main">
|
||||
<span className="capture-text-icon" aria-hidden="true">
|
||||
<svg viewBox="0 0 16 16" fill="none" stroke="currentColor" strokeWidth="1.75" strokeLinecap="round" strokeLinejoin="round" style={{ width: 14, height: 14 }}>
|
||||
<path d="M2 3h12M2 7h12M2 11h8"/>
|
||||
</svg>
|
||||
</span>
|
||||
<div className="capture-text-inputs">
|
||||
<input
|
||||
ref={titleInputRef}
|
||||
className="capture-text-input capture-text-title"
|
||||
type="text"
|
||||
placeholder="Title"
|
||||
value={item.title}
|
||||
onChange={e => onTitleChange(e.target.value)}
|
||||
maxLength={500}
|
||||
/>
|
||||
<textarea
|
||||
className="capture-text-input capture-text-body"
|
||||
placeholder="Body (markdown or plain text)"
|
||||
value={item.body}
|
||||
onChange={e => onBodyChange(e.target.value)}
|
||||
rows={6}
|
||||
/>
|
||||
<div className="capture-text-footer">
|
||||
<select
|
||||
className="capture-text-mime"
|
||||
value={item.mime}
|
||||
onChange={e => onMimeChange(e.target.value)}
|
||||
>
|
||||
<option value="text/markdown">Markdown</option>
|
||||
<option value="text/plain">Plain text</option>
|
||||
</select>
|
||||
</div>
|
||||
</div>
|
||||
<button
|
||||
type="button"
|
||||
className="capture-row-action capture-row-remove"
|
||||
onClick={onRemove}
|
||||
aria-label="Remove"
|
||||
>
|
||||
<svg viewBox="0 0 16 16" fill="none" stroke="currentColor" strokeWidth="2" strokeLinecap="round">
|
||||
<line x1="3" y1="3" x2="13" y2="13"/><line x1="13" y1="3" x2="3" y2="13"/>
|
||||
</svg>
|
||||
</button>
|
||||
</div>
|
||||
</div>
|
||||
)
|
||||
}
|
||||
|
|
|
|||
41
frontend/src/components/CaptureDialog.stories.jsx
Normal file
41
frontend/src/components/CaptureDialog.stories.jsx
Normal file
|
|
@ -0,0 +1,41 @@
|
|||
import { useState } from 'react';
|
||||
import CaptureDialog from './CaptureDialog';
|
||||
|
||||
export default {
|
||||
component: CaptureDialog,
|
||||
tags: ['autodocs'],
|
||||
};
|
||||
|
||||
function CaptureDialogWrapper(args) {
|
||||
const [open, setOpen] = useState(args.open);
|
||||
|
||||
return (
|
||||
<div>
|
||||
<button onClick={() => setOpen(true)} style={{ padding: '8px 16px', marginBottom: '16px' }}>
|
||||
Open Capture Dialog
|
||||
</button>
|
||||
<CaptureDialog
|
||||
{...args}
|
||||
open={open}
|
||||
onClose={() => setOpen(false)}
|
||||
/>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
export const Default = {
|
||||
render: (args) => <CaptureDialogWrapper {...args} />,
|
||||
args: {
|
||||
open: false,
|
||||
archiveId: 'archive_1',
|
||||
onCaptured: () => {},
|
||||
},
|
||||
};
|
||||
|
||||
export const Open = {
|
||||
args: {
|
||||
open: true,
|
||||
archiveId: 'archive_1',
|
||||
onCaptured: () => {},
|
||||
},
|
||||
};
|
||||
|
|
@ -1,44 +1,9 @@
|
|||
import { useState, useEffect, useLayoutEffect, useRef } from 'react'
|
||||
import { fetchEntryTags, assignTag, removeTag, listEntryCollections, listCollections, addEntryToCollection, updateEntryTitle, generateThreadTitle, deleteEntry, rearchiveEntry, pollCaptureJob, fetchEntrySummary, requestEntrySummary, fetchTranscriptionEngines } from '../api'
|
||||
import { useState, useEffect, useRef } from 'react'
|
||||
import { fetchEntryTags, assignTag, removeTag, listEntryCollections, listCollections, addEntryToCollection, updateEntryTitle, deleteEntry, rearchiveEntry, pollCaptureJob } from '../api'
|
||||
import { formatTimestamp, formatBytes, valueText, sourceIconSvg, displayPath } from '../utils'
|
||||
|
||||
const VIS_LABEL = { 0: 'Private', 1: 'Public', 2: 'Users only', 3: 'Public' }
|
||||
|
||||
// Provider labels are display-only; the values are the provider_kind strings
|
||||
// the server persists in entry_summaries.provider_kind.
|
||||
const SUMMARY_PROVIDERS = [
|
||||
{ value: 'anthropic_http', label: 'Anthropic API' },
|
||||
{ value: 'openai_compatible', label: 'OpenAI-compatible API' },
|
||||
{ value: 'claude_cli', label: 'Claude CLI' },
|
||||
{ value: 'codex_cli', label: 'Codex CLI' },
|
||||
]
|
||||
const PROVIDER_LABEL = Object.fromEntries(SUMMARY_PROVIDERS.map(p => [p.value, p.label]))
|
||||
const SUMMARY_PROVIDER_KEY = 'archivr:summary:provider'
|
||||
const SUMMARY_TRANSCRIBE_ENGINE_KEY = 'archivr:summary:transcribe-engine'
|
||||
const SUMMARY_POLL_MS = 1500
|
||||
const UNSUPPORTED_SUMMARY_CONTENT_HEADING = 'This entry can’t be summarized yet.'
|
||||
const UNSUPPORTED_SUMMARY_CONTENT_DETAIL = 'It doesn’t contain archived text that a summary provider can read. Summaries currently support text notes, web pages, X posts and threads, X Articles, and YouTube videos with subtitles. Other video, audio, and image-only entries need a transcript or text source.'
|
||||
const UNSUPPORTED_SUMMARY_CONTENT_MESSAGE = `${UNSUPPORTED_SUMMARY_CONTENT_HEADING}\n\n${UNSUPPORTED_SUMMARY_CONTENT_DETAIL}`
|
||||
|
||||
// Summaries are stored as the raw JSON string the model produced (normalized
|
||||
// server-side to {tldr, summary, tags}). Parsing can still fail for rows written
|
||||
// by an older prompt version, so fall back to showing the text as-is rather than
|
||||
// hiding a summary the user can perfectly well read.
|
||||
function parseSummaryText(text) {
|
||||
if (!text) return null
|
||||
try {
|
||||
const parsed = JSON.parse(text)
|
||||
if (parsed && typeof parsed === 'object') {
|
||||
return {
|
||||
tldr: typeof parsed.tldr === 'string' ? parsed.tldr : '',
|
||||
summary: typeof parsed.summary === 'string' ? parsed.summary : '',
|
||||
tags: Array.isArray(parsed.tags) ? parsed.tags.filter(t => typeof t === 'string') : [],
|
||||
}
|
||||
}
|
||||
} catch { /* not JSON — fall through */ }
|
||||
return { tldr: '', summary: text, tags: [] }
|
||||
}
|
||||
|
||||
|
||||
const ExternalIcon = () => (
|
||||
<svg viewBox="0 0 24 24" fill="none" stroke="currentColor" strokeWidth="2" strokeLinecap="round" strokeLinejoin="round">
|
||||
|
|
@ -58,43 +23,9 @@ export default function ContextRail({ archiveId, selectedEntry, selectedUids, se
|
|||
const [rearchiveState, setRearchiveState] = useState('idle') // 'idle' | 'running' | 'done' | 'error'
|
||||
const [rearchiveError, setRearchiveError] = useState('')
|
||||
const rearchivePollRef = useRef(null)
|
||||
const [titleGenState, setTitleGenState] = useState('idle') // 'idle' | 'running' | 'done' | 'error'
|
||||
const [titleGenError, setTitleGenError] = useState('')
|
||||
const [fontsOpen, setFontsOpen] = useState(false)
|
||||
useEffect(() => { setFontsOpen(false) }, [detail?.summary?.entry_uid])
|
||||
|
||||
// ── Summary state ───────────────────────────────────────────────────────
|
||||
// A completed summary and its replacement attempt are intentionally separate:
|
||||
// regeneration must not blank or overwrite readable content while it runs.
|
||||
const [summary, setSummary] = useState(null)
|
||||
const [summaryAttempt, setSummaryAttempt] = useState(null)
|
||||
const [summaryError, setSummaryError] = useState('')
|
||||
const [summaryBusy, setSummaryBusy] = useState(false)
|
||||
const [summaryProvider, setSummaryProvider] = useState(() => {
|
||||
try {
|
||||
return sessionStorage.getItem(SUMMARY_PROVIDER_KEY) || SUMMARY_PROVIDERS[0].value
|
||||
} catch { return SUMMARY_PROVIDERS[0].value }
|
||||
})
|
||||
const [includeSummaryImages, setIncludeSummaryImages] = useState(false)
|
||||
// Local transcription engines offered by the server ([] = feature off) and
|
||||
// the one picked for YouTube videos without subtitles ('' = none).
|
||||
const [transcriptionEngines, setTranscriptionEngines] = useState([])
|
||||
const [transcribeEngine, setTranscribeEngine] = useState(() => {
|
||||
try {
|
||||
return sessionStorage.getItem(SUMMARY_TRANSCRIBE_ENGINE_KEY) || ''
|
||||
} catch { return '' }
|
||||
})
|
||||
const summaryPollRef = useRef(null)
|
||||
const summaryPollAbortRef = useRef(null)
|
||||
const summaryGenerateAbortRef = useRef(null)
|
||||
const summarySelectionRef = useRef(null)
|
||||
// Update before effects run from the list selection, not detail: detail can
|
||||
// briefly describe the previously selected entry while its replacement loads.
|
||||
const summarySelectionKey = archiveId && selectedEntry?.entry_uid
|
||||
? `${archiveId}:${selectedEntry.entry_uid}`
|
||||
: null
|
||||
summarySelectionRef.current = summarySelectionKey
|
||||
|
||||
// ── Bulk-panel state ────────────────────────────────────────────────────
|
||||
const isBulk = selectedUids?.size >= 2
|
||||
const [bulkTagInput, setBulkTagInput] = useState('')
|
||||
|
|
@ -108,21 +39,6 @@ export default function ContextRail({ archiveId, selectedEntry, selectedUids, se
|
|||
const [singleCollUid, setSingleCollUid] = useState('')
|
||||
const [singleCollState, setSingleCollState] = useState('idle')
|
||||
const [singleCollError, setSingleCollError] = useState('')
|
||||
// Batch thread-title generation; the seq ref invalidates a run whenever the
|
||||
// selection (as a set of uids) or archive changes.
|
||||
const [bulkTitleState, setBulkTitleState] = useState('idle') // 'idle'|'running'|'done'
|
||||
const [bulkTitleProgress, setBulkTitleProgress] = useState({ done: 0, total: 0 })
|
||||
const [bulkTitleSummary, setBulkTitleSummary] = useState({ updated: 0, failed: 0, firstError: '' })
|
||||
const bulkTitleSeqRef = useRef(0)
|
||||
const bulkThreads = (selectedEntries || []).filter(e => e.entity_kind === 'tweet_thread')
|
||||
const bulkSkipped = (selectedUids?.size || 0) - bulkThreads.length
|
||||
const bulkSelectionKey = selectedUids ? [...selectedUids].sort().join(',') : ''
|
||||
useEffect(() => {
|
||||
bulkTitleSeqRef.current++
|
||||
setBulkTitleState('idle')
|
||||
setBulkTitleProgress({ done: 0, total: 0 })
|
||||
setBulkTitleSummary({ updated: 0, failed: 0, firstError: '' })
|
||||
}, [bulkSelectionKey, archiveId])
|
||||
|
||||
useEffect(() => {
|
||||
const seq = ++selectSeqRef.current
|
||||
|
|
@ -154,167 +70,12 @@ export default function ContextRail({ archiveId, selectedEntry, selectedUids, se
|
|||
}).catch(() => {})
|
||||
}, [selectedEntry, archiveId, isPublicSession])
|
||||
|
||||
// F4 state is keyed on entry identity, not the entry object: a successful
|
||||
// title generation replaces `selectedEntry` (new title) and must not reset
|
||||
// its own "Title updated." confirmation.
|
||||
const titleGenSeqRef = useRef(0)
|
||||
const selectedEntryUid = selectedEntry?.entry_uid
|
||||
useEffect(() => {
|
||||
titleGenSeqRef.current++
|
||||
setTitleGenState('idle')
|
||||
setTitleGenError('')
|
||||
}, [selectedEntryUid, archiveId])
|
||||
|
||||
useEffect(() => {
|
||||
return () => {
|
||||
clearInterval(rearchivePollRef.current)
|
||||
}
|
||||
}, [])
|
||||
|
||||
// Seed the summary from the entry detail payload and stop any poll left over
|
||||
// from the previously selected entry.
|
||||
useLayoutEffect(() => {
|
||||
clearInterval(summaryPollRef.current)
|
||||
summaryPollRef.current = null
|
||||
summaryPollAbortRef.current?.abort()
|
||||
summaryPollAbortRef.current = null
|
||||
summaryGenerateAbortRef.current?.abort()
|
||||
summaryGenerateAbortRef.current = null
|
||||
const detailMatchesSelection = selectedEntry?.entry_uid != null &&
|
||||
detail?.summary?.entry_uid === selectedEntry?.entry_uid
|
||||
setSummary(detailMatchesSelection ? detail.latest_summary ?? null : null)
|
||||
setSummaryAttempt(detailMatchesSelection ? detail.summary_attempt ?? null : null)
|
||||
setSummaryError('')
|
||||
setSummaryBusy(false)
|
||||
setIncludeSummaryImages(false)
|
||||
}, [archiveId, selectedEntry?.entry_uid, detail?.summary?.entry_uid])
|
||||
|
||||
// Poll only while a replacement attempt is non-terminal. Anchoring the effect
|
||||
// on its status means a job still running when the user navigates away and
|
||||
// back is picked up again without displacing completed content.
|
||||
const summaryAttemptStatus = summaryAttempt?.status
|
||||
useEffect(() => {
|
||||
clearInterval(summaryPollRef.current)
|
||||
summaryPollRef.current = null
|
||||
if (summaryAttemptStatus !== 'pending' && summaryAttemptStatus !== 'running') return
|
||||
if (!archiveId || !detail?.summary?.entry_uid) return
|
||||
const entryUid = detail.summary.entry_uid
|
||||
const selectionKey = `${archiveId}:${entryUid}`
|
||||
if (summarySelectionKey !== selectionKey) return
|
||||
const controller = new AbortController()
|
||||
summaryPollAbortRef.current = controller
|
||||
const poll = async () => {
|
||||
try {
|
||||
const res = await fetchEntrySummary(archiveId, entryUid, { signal: controller.signal })
|
||||
if (controller.signal.aborted || summarySelectionRef.current !== selectionKey) return
|
||||
setSummary(res.summary ?? null)
|
||||
setSummaryAttempt(res.attempt ?? null)
|
||||
const st = res.attempt?.status
|
||||
if (st !== 'pending' && st !== 'running') {
|
||||
clearInterval(intervalId)
|
||||
if (summaryPollRef.current === intervalId) summaryPollRef.current = null
|
||||
setSummaryBusy(false)
|
||||
if (st === 'completed' && summarySelectionRef.current === selectionKey) onDetailRefresh?.()
|
||||
}
|
||||
} catch (e) {
|
||||
if (controller.signal.aborted || summarySelectionRef.current !== selectionKey) return
|
||||
// A transient poll failure is not worth tearing the section down; the
|
||||
// next tick retries, and a real failure lands as status === 'failed'.
|
||||
}
|
||||
}
|
||||
const intervalId = setInterval(poll, SUMMARY_POLL_MS)
|
||||
summaryPollRef.current = intervalId
|
||||
return () => {
|
||||
clearInterval(intervalId)
|
||||
if (summaryPollRef.current === intervalId) summaryPollRef.current = null
|
||||
controller.abort()
|
||||
if (summaryPollAbortRef.current === controller) summaryPollAbortRef.current = null
|
||||
}
|
||||
}, [summaryAttemptStatus, archiveId, selectedEntry?.entry_uid, detail?.summary?.entry_uid, summarySelectionKey])
|
||||
|
||||
useEffect(() => () => {
|
||||
clearInterval(summaryPollRef.current)
|
||||
summaryPollAbortRef.current?.abort()
|
||||
summaryGenerateAbortRef.current?.abort()
|
||||
}, [])
|
||||
|
||||
useEffect(() => {
|
||||
if (isPublicSession) {
|
||||
setTranscriptionEngines([])
|
||||
return
|
||||
}
|
||||
const controller = new AbortController()
|
||||
fetchTranscriptionEngines({ signal: controller.signal }).then(list => {
|
||||
if (controller.signal.aborted) return
|
||||
const engines = Array.isArray(list) ? list : []
|
||||
setTranscriptionEngines(engines)
|
||||
setTranscribeEngine(current => {
|
||||
if (!current || engines.some(t => t.kind === current)) return current
|
||||
try { sessionStorage.removeItem(SUMMARY_TRANSCRIBE_ENGINE_KEY) } catch { /* private mode */ }
|
||||
return ''
|
||||
})
|
||||
}).catch(() => {
|
||||
if (!controller.signal.aborted) setTranscriptionEngines([])
|
||||
})
|
||||
return () => controller.abort()
|
||||
}, [isPublicSession])
|
||||
|
||||
// The engine is only offered (and sent) for YouTube videos, so a stale
|
||||
// session choice never reaches the server for other entries.
|
||||
const transcribeAvailable = transcriptionEngines.length > 0 &&
|
||||
detail?.summary?.source_kind === 'youtube' &&
|
||||
detail?.summary?.entity_kind === 'video'
|
||||
|
||||
async function handleGenerateSummary(force = false) {
|
||||
if (!archiveId || !detail?.summary?.entry_uid || summaryBusy) return
|
||||
const entryUid = detail.summary.entry_uid
|
||||
const selectionKey = `${archiveId}:${entryUid}`
|
||||
if (summarySelectionRef.current !== selectionKey) return
|
||||
const controller = new AbortController()
|
||||
summaryGenerateAbortRef.current?.abort()
|
||||
summaryGenerateAbortRef.current = controller
|
||||
setSummaryBusy(true)
|
||||
setSummaryError('')
|
||||
try {
|
||||
const res = await requestEntrySummary(archiveId, entryUid, {
|
||||
provider: summaryProvider,
|
||||
force,
|
||||
includeImages: includeSummaryImages,
|
||||
transcribeEngine: transcribeAvailable ? transcribeEngine : '',
|
||||
signal: controller.signal,
|
||||
})
|
||||
if (controller.signal.aborted || summarySelectionRef.current !== selectionKey) return
|
||||
if (res.status === 'completed') {
|
||||
// 200 cache hit: the response *is* the row, no polling needed.
|
||||
setSummary(res)
|
||||
setSummaryAttempt(null)
|
||||
setSummaryBusy(false)
|
||||
if (summarySelectionRef.current === selectionKey) onDetailRefresh?.()
|
||||
} else {
|
||||
// 202: seed a local pending row so the poll effect starts immediately
|
||||
// rather than waiting a tick for the first GET.
|
||||
setSummaryAttempt({ ...(res ?? {}), status: 'pending' })
|
||||
}
|
||||
} catch (e) {
|
||||
if (controller.signal.aborted || summarySelectionRef.current !== selectionKey) return
|
||||
setSummaryError(e.message || 'Summary request failed')
|
||||
setSummaryBusy(false)
|
||||
} finally {
|
||||
if (summaryGenerateAbortRef.current === controller) summaryGenerateAbortRef.current = null
|
||||
}
|
||||
}
|
||||
|
||||
function handleProviderChange(value) {
|
||||
setSummaryProvider(value)
|
||||
if (value === 'claude_cli') setIncludeSummaryImages(false)
|
||||
try { sessionStorage.setItem(SUMMARY_PROVIDER_KEY, value) } catch { /* private mode */ }
|
||||
}
|
||||
|
||||
function handleTranscribeEngineChange(value) {
|
||||
setTranscribeEngine(value)
|
||||
try { sessionStorage.setItem(SUMMARY_TRANSCRIBE_ENGINE_KEY, value) } catch { /* private mode */ }
|
||||
}
|
||||
|
||||
// Fetch available collections whenever archiveId is available
|
||||
useEffect(() => {
|
||||
if (!archiveId) { setCollections([]); return }
|
||||
|
|
@ -506,58 +267,6 @@ export default function ContextRail({ archiveId, selectedEntry, selectedUids, se
|
|||
}
|
||||
}
|
||||
|
||||
async function handleGenerateThreadTitle() {
|
||||
if (!selectedEntry || !archiveId || titleGenState === 'running') return
|
||||
const startSeq = titleGenSeqRef.current
|
||||
const entryUid = selectedEntry.entry_uid
|
||||
setTitleGenState('running')
|
||||
setTitleGenError('')
|
||||
try {
|
||||
const { title } = await generateThreadTitle(archiveId, entryUid, { provider: summaryProvider })
|
||||
// Saved server-side regardless of selection; App's caches are keyed by uid.
|
||||
// Done state first (if still on the same entry), then notify App.
|
||||
if (titleGenSeqRef.current === startSeq) setTitleGenState('done')
|
||||
onEntryTitleChange?.(entryUid, title)
|
||||
} catch (e) {
|
||||
if (titleGenSeqRef.current !== startSeq) return
|
||||
setTitleGenState('error')
|
||||
setTitleGenError(e.message || 'Title generation failed.')
|
||||
}
|
||||
}
|
||||
|
||||
async function handleBulkGenerateTitles() {
|
||||
if (!archiveId || bulkTitleState === 'running' || bulkThreads.length === 0) return
|
||||
const startSeq = ++bulkTitleSeqRef.current
|
||||
const uids = bulkThreads.map(e => e.entry_uid)
|
||||
const provider = summaryProvider
|
||||
let done = 0, updated = 0, failed = 0, firstError = ''
|
||||
setBulkTitleState('running')
|
||||
setBulkTitleProgress({ done: 0, total: uids.length })
|
||||
setBulkTitleSummary({ updated: 0, failed: 0, firstError: '' })
|
||||
let next = 0
|
||||
async function worker() {
|
||||
while (next < uids.length) {
|
||||
if (bulkTitleSeqRef.current !== startSeq) return
|
||||
const uid = uids[next++]
|
||||
try {
|
||||
const { title } = await generateThreadTitle(archiveId, uid, { provider })
|
||||
updated++
|
||||
// Saved server-side; App's caches are keyed by uid, so always propagate.
|
||||
onEntryTitleChange?.(uid, title)
|
||||
} catch (e) {
|
||||
failed++
|
||||
if (!firstError) firstError = e.message || 'Title generation failed.'
|
||||
}
|
||||
done++
|
||||
if (bulkTitleSeqRef.current === startSeq) setBulkTitleProgress({ done, total: uids.length })
|
||||
}
|
||||
}
|
||||
await Promise.all([worker(), worker()])
|
||||
if (bulkTitleSeqRef.current !== startSeq) return
|
||||
setBulkTitleState('done')
|
||||
setBulkTitleSummary({ updated, failed, firstError })
|
||||
}
|
||||
|
||||
const metaRows = detail ? [
|
||||
['Added', formatTimestamp(detail.summary.archived_at)],
|
||||
['Source', detail.summary.source_kind],
|
||||
|
|
@ -567,7 +276,7 @@ export default function ContextRail({ archiveId, selectedEntry, selectedUids, se
|
|||
] : []
|
||||
|
||||
const AUDIO_EXTS = new Set(['mp3','ogg','m4a','opus','wav','flac','aac'])
|
||||
const PREVIEW_EXTS = new Set(['mp4','webm','mov','mkv','avi','m4v','ogv','pdf','html','htm','md','markdown','txt','jpg','jpeg','png','gif','webp','avif','svg','bmp'])
|
||||
const PREVIEW_EXTS = new Set(['mp4','webm','mov','mkv','avi','m4v','ogv','pdf','html','htm','jpg','jpeg','png','gif','webp','avif','svg','bmp'])
|
||||
const primaryMediaIdx = detail ? detail.artifacts.findIndex(a => a.artifact_role === 'primary_media') : -1
|
||||
const primaryMedia = primaryMediaIdx >= 0 ? detail.artifacts[primaryMediaIdx] : null
|
||||
const pmExt = primaryMedia ? primaryMedia.relpath.split('.').pop().toLowerCase() : ''
|
||||
|
|
@ -651,32 +360,6 @@ export default function ContextRail({ archiveId, selectedEntry, selectedUids, se
|
|||
</div>
|
||||
)}
|
||||
|
||||
{bulkThreads.length > 0 && (
|
||||
<div className="rail-section">
|
||||
<div className="rail-section-heading">Thread titles</div>
|
||||
<button
|
||||
className="rail-rearchive-btn"
|
||||
onClick={handleBulkGenerateTitles}
|
||||
disabled={bulkTitleState === 'running'}
|
||||
title={`Names each thread with a small model via ${PROVIDER_LABEL[summaryProvider] || summaryProvider} (the Summary provider)`}
|
||||
>
|
||||
{bulkTitleState === 'running'
|
||||
? `Generating titles\u2026 ${bulkTitleProgress.done}/${bulkTitleProgress.total}`
|
||||
: `Generate titles for ${bulkThreads.length} thread${bulkThreads.length === 1 ? '' : 's'}`}
|
||||
</button>
|
||||
{bulkSkipped > 0 && (
|
||||
<p className="bulk-title-note">
|
||||
{`${bulkSkipped} non-thread entr${bulkSkipped === 1 ? 'y' : 'ies'} will be skipped.`}
|
||||
</p>
|
||||
)}
|
||||
{bulkTitleState === 'done' && (
|
||||
<p className={`form-msg ${bulkTitleSummary.failed ? 'form-msg--err' : 'form-msg--ok'} bulk-title-note`} role="status">
|
||||
{`${bulkTitleSummary.updated} updated${bulkTitleSummary.failed ? `, ${bulkTitleSummary.failed} failed: ${bulkTitleSummary.firstError}` : ''}`}
|
||||
</p>
|
||||
)}
|
||||
</div>
|
||||
)}
|
||||
|
||||
<div className="rail-delete-zone">
|
||||
<button
|
||||
className="rail-delete-btn"
|
||||
|
|
@ -752,123 +435,6 @@ export default function ContextRail({ archiveId, selectedEntry, selectedUids, se
|
|||
</button>
|
||||
)}
|
||||
|
||||
{(() => {
|
||||
// Public sessions get read-only treatment: the completed text if the
|
||||
// server's visibility gate let the detail through at all, and never
|
||||
// the provider selector or Generate button.
|
||||
const parsed = summary?.status === 'completed'
|
||||
? parseSummaryText(summary.summary_text)
|
||||
: null
|
||||
const running = summaryAttempt?.status === 'pending' || summaryAttempt?.status === 'running'
|
||||
const unsupportedContent =
|
||||
(summaryAttempt?.status === 'failed' && summaryAttempt.error_text === UNSUPPORTED_SUMMARY_CONTENT_MESSAGE) ||
|
||||
summaryError === UNSUPPORTED_SUMMARY_CONTENT_MESSAGE
|
||||
if (isPublicSession && !parsed) return null
|
||||
return (
|
||||
<div className="rail-section rail-summary">
|
||||
<div className="rail-section-heading">Summary</div>
|
||||
|
||||
{parsed && (
|
||||
<div className="rail-summary-body">
|
||||
{parsed.tldr && <p className="rail-summary-tldr">{parsed.tldr}</p>}
|
||||
{parsed.summary && <p className="rail-summary-text">{parsed.summary}</p>}
|
||||
{parsed.tags.length > 0 && (
|
||||
<div className="rail-summary-tags">
|
||||
{parsed.tags.map(t => (
|
||||
<span key={t} className="rail-summary-tag">{t}</span>
|
||||
))}
|
||||
</div>
|
||||
)}
|
||||
<p className="rail-summary-provider">
|
||||
{PROVIDER_LABEL[summary.provider_kind] || summary.provider_kind}
|
||||
{summary.resolved_model || summary.provider_model
|
||||
? ` \u00b7 ${summary.resolved_model || summary.provider_model}`
|
||||
: ''}
|
||||
</p>
|
||||
</div>
|
||||
)}
|
||||
|
||||
{running && (
|
||||
<p className="rail-summary-status">
|
||||
<span className="rail-summary-spinner" aria-hidden="true" />
|
||||
{'Generating\u2026'}
|
||||
</p>
|
||||
)}
|
||||
|
||||
{unsupportedContent && !isPublicSession && (
|
||||
<div className="rail-summary-info" role="status">
|
||||
<p className="rail-summary-info__heading">{UNSUPPORTED_SUMMARY_CONTENT_HEADING}</p>
|
||||
<p className="rail-summary-info__detail">{UNSUPPORTED_SUMMARY_CONTENT_DETAIL}</p>
|
||||
</div>
|
||||
)}
|
||||
{summaryAttempt?.status === 'failed' && summaryAttempt.error_text && !unsupportedContent && !isPublicSession && (
|
||||
<p className="form-msg form-msg--err rail-summary-error">
|
||||
{summaryAttempt.error_text}
|
||||
</p>
|
||||
)}
|
||||
{summaryError && !unsupportedContent && (
|
||||
<p className="form-msg form-msg--err rail-summary-error">
|
||||
{summaryError}
|
||||
</p>
|
||||
)}
|
||||
|
||||
{!isPublicSession && !running && (
|
||||
<div className="rail-summary-controls">
|
||||
<select
|
||||
className="rail-summary-select"
|
||||
value={summaryProvider}
|
||||
onChange={e => handleProviderChange(e.target.value)}
|
||||
aria-label="Summary provider"
|
||||
>
|
||||
{SUMMARY_PROVIDERS.map(p => (
|
||||
<option key={p.value} value={p.value}>{p.label}</option>
|
||||
))}
|
||||
</select>
|
||||
{transcribeAvailable && (
|
||||
<>
|
||||
<select
|
||||
className="rail-summary-select"
|
||||
value={transcribeEngine}
|
||||
onChange={e => handleTranscribeEngineChange(e.target.value)}
|
||||
aria-label="Local transcription if no subtitles"
|
||||
>
|
||||
<option value="">No local transcription</option>
|
||||
{transcriptionEngines.map(t => (
|
||||
<option key={t.kind} value={t.kind}>{t.label}{t.english_only ? ' (English only)' : ''}</option>
|
||||
))}
|
||||
</select>
|
||||
<p className="rail-summary-transcribe-note">Used only if this video has no subtitles. Transcription runs on this server and can take several minutes.</p>
|
||||
</>
|
||||
)}
|
||||
<div className={`rail-summary-image-option${summaryProvider === 'claude_cli' ? ' rail-summary-image-option--disabled' : ''}`}>
|
||||
<label className="rail-summary-image-option__label">
|
||||
<input
|
||||
type="checkbox"
|
||||
checked={includeSummaryImages}
|
||||
disabled={summaryProvider === 'claude_cli'}
|
||||
onChange={e => setIncludeSummaryImages(e.target.checked)}
|
||||
/>
|
||||
Include attached images
|
||||
</label>
|
||||
<p className="rail-summary-image-option__note">
|
||||
{summaryProvider === 'claude_cli'
|
||||
? 'Claude CLI cannot attach local images. Choose an HTTP provider or Codex CLI.'
|
||||
: 'Selected archived images are sent to the chosen provider. Up to 4 supported images (5 MiB each, 12 MiB total) can be attached; unsupported or oversized artifacts are skipped.'}
|
||||
</p>
|
||||
</div>
|
||||
<button
|
||||
className="rail-rearchive-btn"
|
||||
onClick={() => handleGenerateSummary(!!parsed)}
|
||||
disabled={summaryBusy}
|
||||
>
|
||||
{summaryBusy ? '\u2026' : parsed ? 'Regenerate' : 'Generate'}
|
||||
</button>
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
)
|
||||
})()}
|
||||
|
||||
<div className="meta-list">
|
||||
{metaRows.filter(([, v]) => v != null && v !== '').map(([label, value]) => (
|
||||
<div key={label} className="meta-item">
|
||||
|
|
@ -1029,25 +595,6 @@ export default function ContextRail({ archiveId, selectedEntry, selectedUids, se
|
|||
{rearchiveState === 'error' && (
|
||||
<p className="form-msg form-msg--err" style={{ marginTop: '6px' }}>{rearchiveError}</p>
|
||||
)}
|
||||
{detail.summary.entity_kind === 'tweet_thread' && (
|
||||
<>
|
||||
<button
|
||||
className="rail-rearchive-btn"
|
||||
style={{ marginTop: '8px' }}
|
||||
onClick={handleGenerateThreadTitle}
|
||||
disabled={titleGenState === 'running'}
|
||||
title={`Names this thread with a small model via ${PROVIDER_LABEL[summaryProvider] || summaryProvider} (the Summary provider)`}
|
||||
>
|
||||
{titleGenState === 'running' ? 'Generating title\u2026' : 'Generate title'}
|
||||
</button>
|
||||
{titleGenState === 'done' && (
|
||||
<p className="form-msg form-msg--ok" style={{ marginTop: '6px' }}>Title updated.</p>
|
||||
)}
|
||||
{titleGenState === 'error' && (
|
||||
<p className="form-msg form-msg--err" style={{ marginTop: '6px' }}>{titleGenError}</p>
|
||||
)}
|
||||
</>
|
||||
)}
|
||||
</div>
|
||||
)}
|
||||
|
||||
|
|
|
|||
|
|
@ -2,7 +2,7 @@ import SkeletonEntryRow from './SkeletonEntryRow';
|
|||
|
||||
import EntryRow from './EntryRow';
|
||||
|
||||
export default function EntriesView({ entries, selectedUids, onRowClick, archiveId, pendingCaptures = [], deletedUids, renamedTitles, isPublicSession, canReorder = false, onChildReorderError }) {
|
||||
export default function EntriesView({ entries, selectedUids, onRowClick, archiveId, pendingCaptures = [], deletedUids, isPublicSession }) {
|
||||
return (
|
||||
<section id="archive-view" className="view is-active">
|
||||
<div className="entry-table">
|
||||
|
|
@ -16,7 +16,7 @@ export default function EntriesView({ entries, selectedUids, onRowClick, archive
|
|||
</div>
|
||||
<div id="entries-body">
|
||||
{pendingCaptures.filter(c => c.archiveId === archiveId).reverse().map(cap => (
|
||||
<SkeletonEntryRow key={cap.id} locator={cap.locator} />
|
||||
<SkeletonEntryRow key={cap.id} />
|
||||
))}
|
||||
{entries.map((entry, idx) => (
|
||||
<EntryRow
|
||||
|
|
@ -29,10 +29,7 @@ export default function EntriesView({ entries, selectedUids, onRowClick, archive
|
|||
onRowClick={onRowClick}
|
||||
selectedUids={selectedUids}
|
||||
deletedUids={deletedUids}
|
||||
renamedTitles={renamedTitles}
|
||||
isPublicSession={isPublicSession}
|
||||
canReorder={canReorder}
|
||||
onReorderError={onChildReorderError}
|
||||
/>
|
||||
))}
|
||||
</div>
|
||||
|
|
|
|||
|
|
@ -1,23 +1,15 @@
|
|||
import { useState, useRef, useEffect } from 'react';
|
||||
import { useState } from 'react';
|
||||
import { formatTimestamp, formatBytes, valueText, sourceIconSvg } from '../utils';
|
||||
import { fetchEntryChildren, reorderEntryChildren } from '../api';
|
||||
import { fetchEntryChildren } from '../api';
|
||||
|
||||
function ChildRow({
|
||||
entry, index, onRowClick, selectedUids,
|
||||
isFirst, isLast, reorderDisabled, onMove, canReorder,
|
||||
isDragging, dropEdge, onHandleDragStart, onHandleDragEnd, onRowDragOver, onRowDrop,
|
||||
}) {
|
||||
function ChildRow({ entry, index, onRowClick, selectedUids }) {
|
||||
const isSelected = (selectedUids?.size === 1) && selectedUids.has(entry.entry_uid);
|
||||
const isMultiSelected = (selectedUids?.size >= 2) && selectedUids.has(entry.entry_uid);
|
||||
const label = valueText(entry.title) || valueText(entry.entry_uid);
|
||||
|
||||
const cls = ['child-entry-row',
|
||||
index % 2 === 0 ? 'child-entry-row--light' : 'child-entry-row--dark',
|
||||
isSelected && 'is-selected',
|
||||
isMultiSelected && 'is-multi-selected',
|
||||
isDragging && 'is-dragging',
|
||||
dropEdge === 'before' && 'is-drop-before',
|
||||
dropEdge === 'after' && 'is-drop-after',
|
||||
].filter(Boolean).join(' ');
|
||||
|
||||
return (
|
||||
|
|
@ -27,54 +19,15 @@ function ChildRow({
|
|||
data-entry-uid={entry.entry_uid}
|
||||
onMouseDown={e => { if (e.shiftKey) e.preventDefault(); }}
|
||||
onClick={e => onRowClick(entry, e)}
|
||||
onDragOver={canReorder ? onRowDragOver : undefined}
|
||||
onDrop={canReorder ? onRowDrop : undefined}
|
||||
aria-keyshortcuts={canReorder ? 'Alt+ArrowUp Alt+ArrowDown' : undefined}
|
||||
onKeyDown={e => {
|
||||
if (canReorder && e.altKey && (e.key === 'ArrowUp' || e.key === 'ArrowDown')) {
|
||||
e.preventDefault();
|
||||
if (!reorderDisabled) onMove(e.key === 'ArrowUp' ? -1 : 1);
|
||||
return;
|
||||
}
|
||||
if (e.key === 'Enter') onRowClick(entry, e);
|
||||
}}
|
||||
onKeyDown={e => { if (e.key === 'Enter') onRowClick(entry, e); }}
|
||||
>
|
||||
<div className="col-check" aria-hidden="true" />
|
||||
<div className="col-added">{formatTimestamp(entry.archived_at)}</div>
|
||||
<div className="col-title">
|
||||
{canReorder && (
|
||||
<span className="child-reorder-controls">
|
||||
<span
|
||||
className="child-drag-handle"
|
||||
draggable={!reorderDisabled}
|
||||
title="Drag to reorder (or Alt+↑/↓)"
|
||||
aria-hidden="true"
|
||||
onClick={e => e.stopPropagation()}
|
||||
onDragStart={onHandleDragStart}
|
||||
onDragEnd={onHandleDragEnd}
|
||||
>⋮⋮</span>
|
||||
<button
|
||||
type="button"
|
||||
className="child-move-btn"
|
||||
aria-label={`Move ${label} up`}
|
||||
disabled={reorderDisabled || isFirst}
|
||||
onClick={e => { e.stopPropagation(); onMove(-1); }}
|
||||
onKeyDown={e => e.stopPropagation()}
|
||||
>↑</button>
|
||||
<button
|
||||
type="button"
|
||||
className="child-move-btn"
|
||||
aria-label={`Move ${label} down`}
|
||||
disabled={reorderDisabled || isLast}
|
||||
onClick={e => { e.stopPropagation(); onMove(1); }}
|
||||
onKeyDown={e => e.stopPropagation()}
|
||||
>↓</button>
|
||||
</span>
|
||||
)}
|
||||
<span className="source-icon">
|
||||
<span dangerouslySetInnerHTML={{ __html: sourceIconSvg(entry.source_kind) }} />
|
||||
</span>
|
||||
<span className="entry-title">{label}</span>
|
||||
<span className="entry-title">{valueText(entry.title) || valueText(entry.entry_uid)}</span>
|
||||
</div>
|
||||
<div className="col-type">
|
||||
<span className="type-pill">{valueText(entry.entity_kind)}</span>
|
||||
|
|
@ -87,16 +40,11 @@ function ChildRow({
|
|||
);
|
||||
}
|
||||
|
||||
export default function EntryRow({ entry, archiveId, rowIndex, isSelected, isMultiSelected, onRowClick, selectedUids, deletedUids, renamedTitles, isPublicSession, canReorder = false, onReorderError }) {
|
||||
export default function EntryRow({ entry, archiveId, rowIndex, isSelected, isMultiSelected, onRowClick, selectedUids, deletedUids, isPublicSession }) {
|
||||
const [favFailed, setFavFailed] = useState(false);
|
||||
const [expanded, setExpanded] = useState(false);
|
||||
const [children, setChildren] = useState(null);
|
||||
const [childrenLoading, setChildrenLoading] = useState(false);
|
||||
const [reorderSaving, setReorderSaving] = useState(false);
|
||||
const [dragUid, setDragUid] = useState(null);
|
||||
const [dropTarget, setDropTarget] = useState(null); // { uid, after: bool }
|
||||
const childListRef = useRef(null);
|
||||
const focusUidRef = useRef(null);
|
||||
|
||||
const showFavicon =
|
||||
entry.source_kind === 'web' &&
|
||||
|
|
@ -146,83 +94,6 @@ export default function EntryRow({ entry, archiveId, rowIndex, isSelected, isMul
|
|||
}
|
||||
}
|
||||
|
||||
// Deleted children are hidden locally and already gone server-side, so the
|
||||
// filtered list is exactly the set the server expects. Renames are applied on
|
||||
// top of the fetched list, since that list is only fetched once per expansion.
|
||||
const visibleChildren = (children ?? [])
|
||||
.filter(c => !deletedUids?.has(c.entry_uid))
|
||||
.map(c => (renamedTitles?.has(c.entry_uid) ? { ...c, title: renamedTitles.get(c.entry_uid) } : c));
|
||||
|
||||
async function commitChildOrder(next) {
|
||||
const prev = children;
|
||||
setChildren(next); // optimistic
|
||||
setReorderSaving(true);
|
||||
try {
|
||||
await reorderEntryChildren(archiveId, entry.entry_uid, next.map(c => c.entry_uid));
|
||||
} catch (err) {
|
||||
setChildren(prev); // revert
|
||||
onReorderError?.(err.message);
|
||||
if (err.status === 400) { // stale set: resync with the server
|
||||
try { setChildren(await fetchEntryChildren(archiveId, entry.entry_uid)); } catch (_) { /* keep reverted list */ }
|
||||
}
|
||||
} finally {
|
||||
setReorderSaving(false);
|
||||
}
|
||||
}
|
||||
|
||||
function moveChild(fromIdx, toIdx) {
|
||||
if (!canReorder || reorderSaving || fromIdx === toIdx || toIdx < 0 || toIdx >= visibleChildren.length) return;
|
||||
const next = visibleChildren.slice();
|
||||
const [moved] = next.splice(fromIdx, 1);
|
||||
next.splice(toIdx, 0, moved);
|
||||
commitChildOrder(next);
|
||||
}
|
||||
|
||||
function handleChildMove(idx, delta) {
|
||||
focusUidRef.current = visibleChildren[idx]?.entry_uid ?? null;
|
||||
moveChild(idx, idx + delta);
|
||||
}
|
||||
|
||||
// Keep keyboard focus on the moved row (React re-inserts keyed nodes,
|
||||
// which can drop focus in some browsers).
|
||||
useEffect(() => {
|
||||
const uid = focusUidRef.current;
|
||||
if (!uid || !childListRef.current) return;
|
||||
focusUidRef.current = null;
|
||||
const row = childListRef.current.querySelector(`[data-entry-uid="${CSS.escape(uid)}"]`);
|
||||
if (row && !row.contains(document.activeElement)) row.focus();
|
||||
}, [children]);
|
||||
|
||||
function handleDragStart(uid, e) {
|
||||
e.stopPropagation();
|
||||
e.dataTransfer.effectAllowed = 'move';
|
||||
e.dataTransfer.setData('text/plain', uid); // Firefox requires data to start a drag
|
||||
const rowEl = e.currentTarget.closest('.child-entry-row');
|
||||
if (rowEl) e.dataTransfer.setDragImage(rowEl, 16, rowEl.offsetHeight / 2);
|
||||
setDragUid(uid);
|
||||
}
|
||||
function handleDragEnd() { setDragUid(null); setDropTarget(null); }
|
||||
function handleRowDragOver(uid, e) {
|
||||
if (!dragUid) return; // foreign drags (files, other parents) are not droppable
|
||||
e.preventDefault();
|
||||
e.dataTransfer.dropEffect = 'move';
|
||||
const rect = e.currentTarget.getBoundingClientRect();
|
||||
const after = e.clientY > rect.top + rect.height / 2;
|
||||
if (dropTarget?.uid !== uid || dropTarget?.after !== after) setDropTarget({ uid, after });
|
||||
}
|
||||
function handleRowDrop(uid, e) {
|
||||
if (!dragUid) return;
|
||||
e.preventDefault();
|
||||
const fromIdx = visibleChildren.findIndex(c => c.entry_uid === dragUid);
|
||||
const targetIdx = visibleChildren.findIndex(c => c.entry_uid === uid);
|
||||
const after = dropTarget?.uid === uid ? dropTarget.after : false;
|
||||
handleDragEnd();
|
||||
if (fromIdx === -1 || targetIdx === -1) return;
|
||||
let toIdx = targetIdx + (after ? 1 : 0);
|
||||
if (fromIdx < toIdx) toIdx -= 1;
|
||||
moveChild(fromIdx, toIdx);
|
||||
}
|
||||
|
||||
const outerClass = [
|
||||
'entry-row-outer',
|
||||
rowIndex % 2 === 0 ? 'entry-row-outer--light' : 'entry-row-outer--dark',
|
||||
|
|
@ -283,32 +154,18 @@ export default function EntryRow({ entry, archiveId, rowIndex, isSelected, isMul
|
|||
{expanded && (
|
||||
<>
|
||||
{childrenLoading && <div className="child-entries-loading">Loading…</div>}
|
||||
<div
|
||||
className="child-entries"
|
||||
ref={childListRef}
|
||||
aria-busy={reorderSaving}
|
||||
aria-label={`${entry.child_count} child entries`}
|
||||
>
|
||||
{visibleChildren.map((child, idx) => (
|
||||
<ChildRow
|
||||
key={child.entry_uid}
|
||||
entry={child}
|
||||
index={idx}
|
||||
onRowClick={onRowClick}
|
||||
selectedUids={selectedUids}
|
||||
isFirst={idx === 0}
|
||||
isLast={idx === visibleChildren.length - 1}
|
||||
reorderDisabled={reorderSaving}
|
||||
canReorder={canReorder}
|
||||
onMove={delta => handleChildMove(idx, delta)}
|
||||
isDragging={dragUid === child.entry_uid}
|
||||
dropEdge={dropTarget?.uid === child.entry_uid && dragUid !== child.entry_uid ? (dropTarget.after ? 'after' : 'before') : null}
|
||||
onHandleDragStart={e => handleDragStart(child.entry_uid, e)}
|
||||
onHandleDragEnd={handleDragEnd}
|
||||
onRowDragOver={e => handleRowDragOver(child.entry_uid, e)}
|
||||
onRowDrop={e => handleRowDrop(child.entry_uid, e)}
|
||||
/>
|
||||
))}
|
||||
<div className="child-entries" aria-label={`${entry.child_count} child entries`}>
|
||||
{children && children
|
||||
.filter(c => !deletedUids?.has(c.entry_uid))
|
||||
.map((child, idx) => (
|
||||
<ChildRow
|
||||
key={child.entry_uid}
|
||||
entry={child}
|
||||
index={idx}
|
||||
onRowClick={onRowClick}
|
||||
selectedUids={selectedUids}
|
||||
/>
|
||||
))}
|
||||
</div>
|
||||
</>
|
||||
)}
|
||||
|
|
|
|||
66
frontend/src/components/EntryRow.stories.jsx
Normal file
66
frontend/src/components/EntryRow.stories.jsx
Normal file
|
|
@ -0,0 +1,66 @@
|
|||
import EntryRow from './EntryRow';
|
||||
|
||||
export default {
|
||||
component: EntryRow,
|
||||
tags: ['autodocs'],
|
||||
};
|
||||
|
||||
const sampleEntry = {
|
||||
entry_uid: 'entry_123',
|
||||
title: 'A Great Article About Web Design',
|
||||
archived_at: '2026-06-27T10:30:00Z',
|
||||
entity_kind: 'page',
|
||||
source_kind: 'web',
|
||||
total_artifact_bytes: 2048576,
|
||||
original_url: 'https://example.com/article',
|
||||
has_favicon: false,
|
||||
};
|
||||
|
||||
const videoEntry = {
|
||||
entry_uid: 'entry_456',
|
||||
title: 'React Performance Tips',
|
||||
archived_at: '2026-06-26T14:15:00Z',
|
||||
entity_kind: 'video',
|
||||
source_kind: 'youtube',
|
||||
total_artifact_bytes: 104857600,
|
||||
original_url: 'https://youtube.com/watch?v=xyz',
|
||||
has_favicon: false,
|
||||
};
|
||||
|
||||
export const Default = {
|
||||
args: {
|
||||
entry: sampleEntry,
|
||||
archiveId: 'archive_1',
|
||||
isSelected: false,
|
||||
onSelect: () => {},
|
||||
},
|
||||
decorators: [
|
||||
(Story) => (
|
||||
<div style={{
|
||||
display: 'grid',
|
||||
gridTemplateColumns: '178px 38% 130px 110px 34%',
|
||||
gap: '10px',
|
||||
padding: '10px',
|
||||
background: 'var(--paper-3)',
|
||||
}}>
|
||||
<Story />
|
||||
</div>
|
||||
),
|
||||
],
|
||||
};
|
||||
|
||||
export const Selected = {
|
||||
args: {
|
||||
...Default.args,
|
||||
isSelected: true,
|
||||
},
|
||||
decorators: Default.decorators,
|
||||
};
|
||||
|
||||
export const Video = {
|
||||
args: {
|
||||
...Default.args,
|
||||
entry: videoEntry,
|
||||
},
|
||||
decorators: Default.decorators,
|
||||
};
|
||||
|
|
@ -2,12 +2,10 @@ import VideoPreview from './VideoPreview';
|
|||
import IframePreview from './IframePreview';
|
||||
import ImagePreview from './ImagePreview';
|
||||
import TweetPreview from './TweetPreview';
|
||||
import TextPreview from './TextPreview';
|
||||
|
||||
const VIDEO_EXTS = new Set(['mp4', 'webm', 'mov', 'mkv', 'avi', 'm4v', 'ogv']);
|
||||
const AUDIO_EXTS = new Set(['mp3', 'ogg', 'm4a', 'opus', 'wav', 'flac', 'aac']);
|
||||
const IMAGE_EXTS = new Set(['jpg', 'jpeg', 'png', 'gif', 'webp', 'avif', 'svg', 'bmp']);
|
||||
const TEXT_EXTS = new Set(['md', 'markdown', 'txt']);
|
||||
const CONTENT_TYPES = {
|
||||
mp4: 'video/mp4', webm: 'video/webm', mov: 'video/quicktime',
|
||||
mkv: 'video/x-matroska', avi: 'video/x-msvideo', m4v: 'video/mp4', ogv: 'video/ogg',
|
||||
|
|
@ -133,20 +131,7 @@ export default function PreviewPanel({ archiveId, entry, detail, fullPage, onXAr
|
|||
);
|
||||
}
|
||||
|
||||
// 6b. Plain text / Markdown
|
||||
if (TEXT_EXTS.has(ext)) {
|
||||
return (
|
||||
<div className="preview-panel" style={{ flex: 1, minHeight: 0, overflow: 'auto' }}>
|
||||
<TextPreview
|
||||
src={primaryMediaUrl}
|
||||
mime={primaryArtifact.mime_type}
|
||||
title={summary.title}
|
||||
/>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
// 7. Image
|
||||
// 7. Image
|
||||
if (IMAGE_EXTS.has(ext)) {
|
||||
return (
|
||||
<div className="preview-panel" style={{ height: '100%' }}>
|
||||
|
|
|
|||
|
|
@ -1,4 +1,4 @@
|
|||
import { useState, useEffect, useContext, useCallback, useRef } from 'react'
|
||||
import { useState, useEffect, useContext, useCallback } from 'react'
|
||||
import { AuthContext } from '../App.jsx'
|
||||
import {
|
||||
updateProfile, changePassword, patchMe,
|
||||
|
|
@ -6,17 +6,13 @@ import {
|
|||
getInstanceSettings, updateInstanceSettings,
|
||||
scanOrphanBlobs, deleteOrphanBlobs,
|
||||
listCookieRules, createCookieRule, updateCookieRule, deleteCookieRule,
|
||||
listRoles, fetchMe,
|
||||
getYtDlpStatus, updateYtDlp,
|
||||
} from '../api.js'
|
||||
|
||||
const ROLE_ADMIN = 4
|
||||
const ROLE_OWNER = 8
|
||||
|
||||
export default function SettingsView({ tab, onTabChange, archiveId }) {
|
||||
const { currentUser, setCurrentUser } = useContext(AuthContext) ?? {}
|
||||
const isAdmin = currentUser && ((currentUser.role_bits & ROLE_ADMIN) !== 0)
|
||||
const isOwner = !!currentUser && (currentUser.role_bits & ROLE_OWNER) !== 0
|
||||
|
||||
const tabs = ['profile', 'tokens', ...(isAdmin ? ['instance', 'cookies', 'extensions', 'storage'] : [])]
|
||||
const tabLabels = { profile: 'Profile', tokens: 'API Tokens', instance: 'Instance', cookies: 'Cookies', extensions: 'Extensions', storage: 'Storage' }
|
||||
|
|
@ -36,12 +32,7 @@ export default function SettingsView({ tab, onTabChange, archiveId }) {
|
|||
|
||||
{tab === 'profile' && <ProfileTab currentUser={currentUser} setCurrentUser={setCurrentUser} />}
|
||||
{tab === 'tokens' && <TokensTab />}
|
||||
{tab === 'instance' && isAdmin && (
|
||||
<>
|
||||
<InstanceTab isOwner={isOwner} setCurrentUser={setCurrentUser} />
|
||||
<YtDlpSection />
|
||||
</>
|
||||
)}
|
||||
{tab === 'instance' && isAdmin && <InstanceTab />}
|
||||
{tab === 'cookies' && isAdmin && <CookiesTab />}
|
||||
{tab === 'extensions' && isAdmin && <ExtensionsTab />}
|
||||
{tab === 'storage' && isAdmin && <StorageTab archiveId={archiveId} />}
|
||||
|
|
@ -248,64 +239,26 @@ function TokensTab() {
|
|||
)
|
||||
}
|
||||
|
||||
const TITLE_MODEL_PROVIDERS = [
|
||||
['anthropic_http', 'Anthropic API'],
|
||||
['openai_compatible', 'OpenAI-compatible API'],
|
||||
['claude_cli', 'Claude CLI'],
|
||||
['codex_cli', 'Codex CLI'],
|
||||
]
|
||||
|
||||
function InstanceTab({ isOwner, setCurrentUser }) {
|
||||
function InstanceTab() {
|
||||
const [settings, setSettings] = useState(null)
|
||||
const [loading, setLoading] = useState(true)
|
||||
const [error, setError] = useState(null)
|
||||
const [saving, setSaving] = useState(false)
|
||||
const [saveMsg, setSaveMsg] = useState(null)
|
||||
const [roles, setRoles] = useState([])
|
||||
const [reorderBits, setReorderBits] = useState(12)
|
||||
const [permSaving, setPermSaving] = useState(false)
|
||||
const [permMsg, setPermMsg] = useState(null)
|
||||
|
||||
useEffect(() => {
|
||||
(async () => {
|
||||
try {
|
||||
const [s, r] = await Promise.all([getInstanceSettings(), listRoles()])
|
||||
setSettings(s)
|
||||
setRoles(r.filter(role => role.bit_position > 0))
|
||||
setReorderBits(s.reorder_children_role_bits ?? 12)
|
||||
}
|
||||
try { setSettings(await getInstanceSettings()) }
|
||||
catch (e) { setError(e.message) }
|
||||
finally { setLoading(false) }
|
||||
})()
|
||||
}, [])
|
||||
|
||||
function toggleRole(bit, checked) {
|
||||
setReorderBits(b => checked ? (b | bit) >>> 0 : (b & ~bit) >>> 0)
|
||||
}
|
||||
|
||||
async function handleSavePermissions(e) {
|
||||
e.preventDefault()
|
||||
setPermSaving(true); setPermMsg(null)
|
||||
try {
|
||||
await updateInstanceSettings({ reorder_children_role_bits: reorderBits })
|
||||
setSettings(s => ({ ...s, reorder_children_role_bits: reorderBits }))
|
||||
const me = await fetchMe()
|
||||
if (me) setCurrentUser?.(me) // owner's own can_reorder_children may change
|
||||
setPermMsg({ ok: true, text: 'Saved.' })
|
||||
} catch (err) {
|
||||
setPermMsg({ ok: false, text: err.message })
|
||||
} finally {
|
||||
setPermSaving(false)
|
||||
}
|
||||
}
|
||||
|
||||
async function handleSave(e) {
|
||||
e.preventDefault()
|
||||
setSaving(true); setSaveMsg(null)
|
||||
try {
|
||||
const { reorder_children_role_bits: _mask, title_models: _models, ...rest } = settings
|
||||
await updateInstanceSettings(rest)
|
||||
setSettings(await getInstanceSettings()) // server-trimmed models + effective sources
|
||||
await updateInstanceSettings(settings)
|
||||
setSaveMsg({ ok: true, text: 'Saved.' })
|
||||
} catch (err) {
|
||||
setSaveMsg({ ok: false, text: err.message })
|
||||
|
|
@ -343,54 +296,12 @@ function InstanceTab({ isOwner, setCurrentUser }) {
|
|||
<option value={3}>Public</option>
|
||||
</select>
|
||||
</div>
|
||||
<div className="form-field" style={{ marginTop: 4 }}>
|
||||
<label className="form-label">Thread title models</label>
|
||||
{TITLE_MODEL_PROVIDERS.map(([kind, label]) => {
|
||||
const key = `title_model_${kind}`
|
||||
const info = settings.title_models?.[kind]
|
||||
const placeholder = info ? `${info.fallback_model} (${info.fallback_source})` : ''
|
||||
return (
|
||||
<div key={kind} className="form-field">
|
||||
<label className="form-label" htmlFor={key}>{label}</label>
|
||||
<input id={key} className="field-input" type="text" maxLength={100}
|
||||
value={settings[key] ?? ''} placeholder={placeholder}
|
||||
onChange={e => setSettings(s => ({ ...s, [key]: e.target.value }))} />
|
||||
</div>
|
||||
)
|
||||
})}
|
||||
<p className="form-hint">Cheap model used for Generate title. Leave blank to use the default.</p>
|
||||
</div>
|
||||
{saveMsg && <div className={`form-msg form-msg--${saveMsg.ok ? 'ok' : 'err'}`}>{saveMsg.text}</div>}
|
||||
<button className="btn-primary" type="submit" disabled={saving}>
|
||||
{saving ? 'Saving\u2026' : 'Save Settings'}
|
||||
</button>
|
||||
</form>
|
||||
</div>
|
||||
<div className="form-section">
|
||||
<h2>Permissions</h2>
|
||||
<form onSubmit={handleSavePermissions}>
|
||||
<label className="form-label">Reorder child entries</label>
|
||||
{roles.map(role => {
|
||||
const bit = (1 << role.bit_position) >>> 0
|
||||
return (
|
||||
<label key={role.role_uid} className="checkbox-row">
|
||||
<input type="checkbox" disabled={!isOwner || permSaving}
|
||||
checked={(reorderBits & bit) !== 0}
|
||||
onChange={e => toggleRole(bit, e.target.checked)} />
|
||||
{role.name}{!role.is_builtin && <span className="muted"> (custom)</span>}
|
||||
</label>
|
||||
)
|
||||
})}
|
||||
<p className="form-hint">Roles are cumulative: every signed-in account also has User, and owners also have Admin. Checking User lets every signed-in account reorder.</p>
|
||||
{!isOwner && <p className="form-hint">Only the owner can change this.</p>}
|
||||
{permMsg && <div className={`form-msg form-msg--${permMsg.ok ? 'ok' : 'err'}`}>{permMsg.text}</div>}
|
||||
{isOwner && (
|
||||
<button className="btn-primary" type="submit" disabled={permSaving}>
|
||||
{permSaving ? 'Saving\u2026' : 'Save Permissions'}
|
||||
</button>
|
||||
)}
|
||||
</form>
|
||||
</div>
|
||||
</div>
|
||||
)
|
||||
}
|
||||
|
|
@ -890,212 +801,4 @@ function ExtensionsTab() {
|
|||
</div>
|
||||
</div>
|
||||
)
|
||||
}
|
||||
|
||||
const YT_DLP_SOURCE_LABELS = {
|
||||
force: 'Forced (ARCHIVR_YT_DLP_FORCE)',
|
||||
env: 'Pinned (ARCHIVR_YT_DLP)',
|
||||
'state-dir': 'Managed install',
|
||||
path: 'System PATH',
|
||||
}
|
||||
const JS_SOURCE_LABELS = {
|
||||
force: 'Forced (ARCHIVR_JS_RUNTIME)',
|
||||
env: 'Pinned (ARCHIVR_DENO)',
|
||||
'state-dir': 'Managed install',
|
||||
path: 'System PATH',
|
||||
}
|
||||
|
||||
function sourceLabel(labels, row) {
|
||||
return (row?.role && labels[row.role]) || row?.label || 'unknown source'
|
||||
}
|
||||
|
||||
function YtDlpCandidateTable({ caption, rows, labels }) {
|
||||
return (
|
||||
<div className="form-field">
|
||||
<div className="form-label">{caption}</div>
|
||||
<div className="ytdlp-table-wrap">
|
||||
<table className="admin-table ytdlp-table">
|
||||
<thead>
|
||||
<tr><th>Source</th><th>Path</th><th>Version</th><th></th></tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
{rows.map(row => (
|
||||
<tr key={row.role} className={row.chosen ? 'ytdlp-row--chosen' : undefined}>
|
||||
<td>{sourceLabel(labels, row)}</td>
|
||||
<td className="ytdlp-path">
|
||||
{row.path ?? <span className="muted">not set</span>}
|
||||
</td>
|
||||
<td>
|
||||
{row.invalid
|
||||
? <span className="form-msg--err">invalid: {row.invalid}</span>
|
||||
: (row.path ? (row.version ?? '\u2014') : '\u2014')}
|
||||
</td>
|
||||
<td>{row.chosen && <span className="ytdlp-badge">in use</span>}</td>
|
||||
</tr>
|
||||
))}
|
||||
</tbody>
|
||||
</table>
|
||||
</div>
|
||||
</div>
|
||||
)
|
||||
}
|
||||
|
||||
function YtDlpSection() {
|
||||
const [status, setStatus] = useState(null)
|
||||
const [loading, setLoading] = useState(true)
|
||||
const [error, setError] = useState(null)
|
||||
const [updating, setUpdating] = useState(false)
|
||||
const [result, setResult] = useState(null)
|
||||
const [updateError, setUpdateError] = useState(null)
|
||||
|
||||
const loadCtrl = useRef(null)
|
||||
const alive = useRef(true)
|
||||
|
||||
// Probing every candidate takes seconds, so this loads separately from the settings form.
|
||||
// A newer load or unmounting aborts the previous request. Resolves to the status or null.
|
||||
async function load() {
|
||||
loadCtrl.current?.abort()
|
||||
const ctrl = new AbortController()
|
||||
loadCtrl.current = ctrl
|
||||
setLoading(true)
|
||||
setError(null)
|
||||
try {
|
||||
const s = await getYtDlpStatus({ signal: ctrl.signal })
|
||||
if (!ctrl.signal.aborted) setStatus(s)
|
||||
return s
|
||||
} catch (e) {
|
||||
if (!ctrl.signal.aborted) setError(e.message)
|
||||
return null
|
||||
} finally {
|
||||
if (!ctrl.signal.aborted) setLoading(false)
|
||||
}
|
||||
}
|
||||
|
||||
useEffect(() => {
|
||||
alive.current = true
|
||||
load()
|
||||
return () => {
|
||||
alive.current = false
|
||||
loadCtrl.current?.abort()
|
||||
}
|
||||
}, [])
|
||||
|
||||
async function handleUpdate() {
|
||||
setUpdating(true)
|
||||
setResult(null)
|
||||
setUpdateError(null)
|
||||
try {
|
||||
const r = await updateYtDlp()
|
||||
setResult(r)
|
||||
setStatus(r.status)
|
||||
} catch (err) {
|
||||
// A proxy may time out (e.g. 504) while the update keeps running server-side.
|
||||
const s = err.status !== 409 && alive.current ? await load() : null
|
||||
setUpdateError(s?.update_running
|
||||
? 'Update still running on the server — refresh in a minute.'
|
||||
: err.message)
|
||||
} finally {
|
||||
setUpdating(false)
|
||||
}
|
||||
}
|
||||
|
||||
const ytChosen = status?.yt_dlp_chosen
|
||||
const ytChosenRow = (status?.yt_dlp ?? []).find(r => r.chosen)
|
||||
const jsChosen = status?.js_runtime_chosen
|
||||
const inUse = status?.js_runtime_in_use
|
||||
const invalidRows = (status?.js_runtime ?? []).filter(r => r.invalid)
|
||||
|
||||
return (
|
||||
<div className="form-section ytdlp-section">
|
||||
<h2>yt-dlp</h2>
|
||||
{loading && !status && <div className="muted">Loading{'\u2026'}</div>}
|
||||
{error && <div className="form-msg form-msg--err">{error}</div>}
|
||||
|
||||
{status && (
|
||||
<>
|
||||
<dl className="ytdlp-summary">
|
||||
<dt>yt-dlp</dt>
|
||||
<dd>
|
||||
{ytChosen?.version ?? 'unknown'}
|
||||
<span className="muted"> · {ytChosen?.role ? YT_DLP_SOURCE_LABELS[ytChosen.role] : 'unlisted path'}</span>
|
||||
{ytChosen?.path && <div className="ytdlp-path">{ytChosen.path}</div>}
|
||||
</dd>
|
||||
<dt>JS runtime</dt>
|
||||
<dd>
|
||||
{jsChosen ? (
|
||||
<>
|
||||
{jsChosen.kind} {jsChosen.version ?? ''}
|
||||
<span className="muted"> · {JS_SOURCE_LABELS[jsChosen.role] ?? jsChosen.role}</span>
|
||||
{jsChosen.path && <div className="ytdlp-path">{jsChosen.path}</div>}
|
||||
</>
|
||||
) : <span className="form-msg--err">none</span>}
|
||||
</dd>
|
||||
<dt>In use by this server</dt>
|
||||
<dd>
|
||||
{inUse ? (
|
||||
<>
|
||||
{inUse.kind}{jsChosen && jsChosen.path === inUse.path && jsChosen.version ? ` ${jsChosen.version}` : ''}
|
||||
{inUse.path && <div className="ytdlp-path">{inUse.path}</div>}
|
||||
</>
|
||||
) : <span className="muted">no JS runtime</span>}
|
||||
</dd>
|
||||
</dl>
|
||||
|
||||
{ytChosenRow?.invalid && (
|
||||
<div className="form-msg form-msg--err">
|
||||
The yt-dlp in use does not run: {ytChosenRow.invalid}
|
||||
</div>
|
||||
)}
|
||||
{!jsChosen && (
|
||||
<div className="form-msg form-msg--err">
|
||||
No JS runtime resolved — YouTube downloads may fail with HTTP 403. Update below to install Deno.
|
||||
</div>
|
||||
)}
|
||||
{invalidRows.map(r => (
|
||||
<div key={r.role} className="form-msg form-msg--err">
|
||||
ARCHIVR_JS_RUNTIME={r.path} is invalid ({r.invalid}) and is ignored.
|
||||
</div>
|
||||
))}
|
||||
|
||||
{status.state_dir && <p className="form-hint">State directory: <span className="ytdlp-path">{status.state_dir}</span></p>}
|
||||
{!status.yt_dlp_installed && status.yt_dlp_target && (
|
||||
<p className="form-hint">No managed yt-dlp yet — the update installs it to {status.yt_dlp_target}.</p>
|
||||
)}
|
||||
{!status.deno_installed && status.deno_target && (
|
||||
<p className="form-hint">No managed Deno yet — the update installs it to {status.deno_target}.</p>
|
||||
)}
|
||||
|
||||
<YtDlpCandidateTable caption="yt-dlp candidates" rows={status.yt_dlp} labels={YT_DLP_SOURCE_LABELS} />
|
||||
<YtDlpCandidateTable caption="JS runtime candidates" rows={status.js_runtime} labels={JS_SOURCE_LABELS} />
|
||||
</>
|
||||
)}
|
||||
|
||||
{(status || !loading) && (
|
||||
<div className="ytdlp-actions">
|
||||
<button className="btn-primary" type="button"
|
||||
disabled={updating || status?.update_running} onClick={handleUpdate}>
|
||||
{updating ? 'Updating\u2026 (can take a few minutes)' : 'Update yt-dlp & Deno'}
|
||||
</button>
|
||||
<button className="btn-ghost" type="button" disabled={updating || loading} onClick={load}>
|
||||
Refresh
|
||||
</button>
|
||||
{status?.update_running && !updating && <span className="form-hint">An update is already running.</span>}
|
||||
</div>
|
||||
)}
|
||||
|
||||
{result && (
|
||||
<>
|
||||
{[['yt-dlp', result.yt_dlp], ['Deno', result.deno]].map(([name, o]) => (
|
||||
<div key={name} className={`form-msg form-msg--${o.ok ? 'ok' : 'err'}`}>
|
||||
{name}: {o.ok ? o.message : `failed — ${o.message}`}
|
||||
</div>
|
||||
))}
|
||||
{(result.yt_dlp.ok || result.deno.ok) && (
|
||||
<p className="form-hint">New binaries are used for the next capture — no restart needed.</p>
|
||||
)}
|
||||
</>
|
||||
)}
|
||||
{updateError && <div className="form-msg form-msg--err">{updateError}</div>}
|
||||
</div>
|
||||
)
|
||||
}
|
||||
|
|
@ -1,37 +1,23 @@
|
|||
const COLLECTION_MARKERS = [
|
||||
'list=',
|
||||
'/playlist/',
|
||||
'/channel/',
|
||||
'yt:playlist:',
|
||||
'yt:channel:',
|
||||
'ytm:playlist:',
|
||||
'spotify:playlist:',
|
||||
'spotify:album:',
|
||||
];
|
||||
|
||||
function isCollectionLocator(locator) {
|
||||
const normalized = locator.toLowerCase();
|
||||
return COLLECTION_MARKERS.some(marker => normalized.includes(marker));
|
||||
}
|
||||
|
||||
function truncateLocator(locator) {
|
||||
return locator.length > 80 ? `${locator.slice(0, 79)}…` : locator;
|
||||
}
|
||||
|
||||
export default function SkeletonEntryRow({ locator = '' }) {
|
||||
const locatorText = String(locator);
|
||||
const isCollection = isCollectionLocator(locatorText);
|
||||
|
||||
export default function SkeletonEntryRow() {
|
||||
return (
|
||||
<div className="in-progress-entry-row" role="status" aria-live="polite">
|
||||
<span className="cap-spinner in-progress-entry-row__spinner" aria-hidden="true" />
|
||||
<span className="in-progress-entry-row__locator" title={locatorText}>
|
||||
{truncateLocator(locatorText)}
|
||||
</span>
|
||||
<span className="in-progress-entry-row__status">
|
||||
Archiving…
|
||||
{isCollection && <span className="in-progress-entry-row__kind">(playlist)</span>}
|
||||
</span>
|
||||
<div className="skeleton-row">
|
||||
<div className="col-check" aria-hidden="true" />
|
||||
<div className="col-added">
|
||||
<span className="skeleton-cell" style={{ width: 108, height: 13 }} />
|
||||
</div>
|
||||
<div className="col-title" style={{ gap: '0.42em', display: 'flex', alignItems: 'center' }}>
|
||||
<span className="skeleton-cell" style={{ width: 14, height: 14, borderRadius: '50%', flexShrink: 0 }} />
|
||||
<span className="skeleton-cell" style={{ width: '58%', height: 13 }} />
|
||||
</div>
|
||||
<div className="col-type">
|
||||
<span className="skeleton-cell" style={{ width: 58, height: 20, borderRadius: 99 }} />
|
||||
</div>
|
||||
<div className="col-size">
|
||||
<span className="skeleton-cell" style={{ width: 44, height: 12 }} />
|
||||
</div>
|
||||
<div className="col-url">
|
||||
<span className="skeleton-cell" style={{ width: '65%', height: 12 }} />
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
)
|
||||
}
|
||||
|
|
|
|||
|
|
@ -1,43 +0,0 @@
|
|||
import { describe, expect, test } from 'bun:test';
|
||||
import { renderToStaticMarkup } from 'react-dom/server';
|
||||
|
||||
import SkeletonEntryRow from './SkeletonEntryRow';
|
||||
|
||||
describe('SkeletonEntryRow', () => {
|
||||
test('renders a compact archiving status row for a locator', () => {
|
||||
const markup = renderToStaticMarkup(
|
||||
<SkeletonEntryRow locator="tweet:1891234567890123456" />,
|
||||
);
|
||||
|
||||
expect(markup).toContain('in-progress-entry-row__spinner');
|
||||
expect(markup).toContain('tweet:1891234567890123456');
|
||||
expect(markup).toContain('Archiving…');
|
||||
expect(markup).not.toContain('(playlist)');
|
||||
});
|
||||
|
||||
test('labels playlist and channel locators', () => {
|
||||
const collectionLocators = [
|
||||
'https://www.youtube.com/watch?v=abc123&list=PL123',
|
||||
'https://www.youtube.com/playlist/PL123',
|
||||
'https://www.youtube.com/channel/UC123',
|
||||
'yt:playlist:PL123',
|
||||
'yt:channel:UC123',
|
||||
'ytm:playlist:PL123',
|
||||
'spotify:playlist:123',
|
||||
'spotify:album:123',
|
||||
];
|
||||
|
||||
for (const locator of collectionLocators) {
|
||||
const markup = renderToStaticMarkup(<SkeletonEntryRow locator={locator} />);
|
||||
expect(markup).toContain('(playlist)');
|
||||
}
|
||||
});
|
||||
|
||||
test('truncates long locators to 80 displayed characters', () => {
|
||||
const locator = 'x'.repeat(100);
|
||||
const markup = renderToStaticMarkup(<SkeletonEntryRow locator={locator} />);
|
||||
const visibleLocator = markup.match(/in-progress-entry-row__locator"[^>]*>(.*?)<\/span>/)?.[1];
|
||||
|
||||
expect(visibleLocator).toBe(`${'x'.repeat(79)}…`);
|
||||
});
|
||||
});
|
||||
167
frontend/src/components/TagsView.stories.jsx
Normal file
167
frontend/src/components/TagsView.stories.jsx
Normal file
|
|
@ -0,0 +1,167 @@
|
|||
import { useState, useEffect } from 'react';
|
||||
import TagsView from './TagsView';
|
||||
|
||||
// Installs a per-story fetch stub that intercepts /api/ tag mutations and
|
||||
// returns realistic Tag shapes so renameTag/moveTag/createTag don't throw.
|
||||
// Installed in useEffect so it never leaks into other stories and won't
|
||||
// double-wrap on re-renders.
|
||||
function ApiStub({ children }) {
|
||||
useEffect(() => {
|
||||
const realFetch = window.fetch.bind(window);
|
||||
window.fetch = (url, opts) => {
|
||||
if (typeof url === 'string' && url.startsWith('/api/')) {
|
||||
const method = (opts?.method ?? 'GET').toUpperCase();
|
||||
if (method === 'DELETE') {
|
||||
return Promise.resolve(new Response(null, { status: 204 }));
|
||||
}
|
||||
// Parse request body to construct a realistic Tag shape.
|
||||
// renameTag sends { name }, moveTag/createTag send path info.
|
||||
// full_path must be present or callers throw on updated.full_path.
|
||||
let body = {};
|
||||
try { body = JSON.parse(opts?.body ?? '{}'); } catch { /* ignore */ }
|
||||
const slug = (body.name ?? body.path ?? 'stub')
|
||||
.trim().replace(/\s+/g, '-').replace(/[^a-zA-Z0-9-]/g, '').replace(/^-+|-+$/g, '') || 'stub';
|
||||
const stubTag = {
|
||||
tag_uid: 'stub-uid',
|
||||
name: slug.replace(/-/g, ' ').replace(/\b\w/g, c => c.toUpperCase()),
|
||||
slug,
|
||||
full_path: `/${slug}`,
|
||||
};
|
||||
return Promise.resolve(
|
||||
new Response(JSON.stringify(stubTag), {
|
||||
status: method === 'POST' ? 201 : 200,
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
})
|
||||
);
|
||||
}
|
||||
return realFetch(url, opts);
|
||||
};
|
||||
return () => { window.fetch = realFetch; };
|
||||
}, []);
|
||||
return children;
|
||||
}
|
||||
|
||||
function withApiStub(Story) {
|
||||
return <ApiStub><Story /></ApiStub>;
|
||||
}
|
||||
|
||||
export default {
|
||||
component: TagsView,
|
||||
tags: ['autodocs'],
|
||||
parameters: { layout: 'padded' },
|
||||
decorators: [withApiStub],
|
||||
};
|
||||
|
||||
// ── Shared fixtures ───────────────────────────────────────────────────────
|
||||
|
||||
const noop = () => {};
|
||||
|
||||
function tag(tag_uid, name, slug, full_path, entry_count = 0, children = [], subtree_count = null) {
|
||||
return { tag: { tag_uid, name, slug, full_path }, entry_count, subtree_count: subtree_count ?? entry_count, children };
|
||||
}
|
||||
|
||||
const sampleTree = [
|
||||
tag('t1', 'Science', 'science', '/science', 12, [
|
||||
tag('t2', 'Computer Science', 'computer-science', '/science/computer-science', 7, [
|
||||
tag('t3', 'Algorithms', 'algorithms', '/science/computer-science/algorithms', 3),
|
||||
tag('t4', 'Compilers', 'compilers', '/science/computer-science/compilers', 1),
|
||||
], 11),
|
||||
tag('t5', 'Physics', 'physics', '/science/physics', 4),
|
||||
], 23),
|
||||
tag('t6', 'History', 'history', '/history', 5, [
|
||||
tag('t7', 'Ancient', 'ancient', '/history/ancient', 2),
|
||||
], 7),
|
||||
tag('t8', 'Reading List', 'reading-list', '/reading-list', 0),
|
||||
];
|
||||
|
||||
// Wrapper that wires local state so onTagsRefresh/onTagRenamed callbacks
|
||||
// keep the tree consistent within a story session.
|
||||
function TagsViewSandbox({ initialNodes = sampleTree, tagFilter = null, humanizeTags = false }) {
|
||||
const [nodes, setNodes] = useState(initialNodes);
|
||||
const [filter, setFilter] = useState(tagFilter);
|
||||
|
||||
function handleTagRenamed(oldPath, newPath) {
|
||||
if (filter === oldPath) setFilter(newPath);
|
||||
else if (filter?.startsWith(oldPath + '/')) setFilter(newPath + filter.slice(oldPath.length));
|
||||
}
|
||||
|
||||
function handleTagDeleted(deletedPath) {
|
||||
if (filter === deletedPath || filter?.startsWith(deletedPath + '/')) setFilter(null);
|
||||
}
|
||||
|
||||
// onTagsRefresh is a no-op here: in a real app it re-fetches; the stub
|
||||
// tag returned from fetch won't match our fixture tree, so the tree stays
|
||||
// as-is after mutations. That is acceptable for visual QA purposes.
|
||||
|
||||
return (
|
||||
<div style={{ maxWidth: 340, fontFamily: 'Helvetica Neue, sans-serif' }}>
|
||||
<TagsView
|
||||
archiveId="demo"
|
||||
tagNodes={nodes}
|
||||
tagFilter={filter}
|
||||
onTagFilterSet={setFilter}
|
||||
onViewChange={noop}
|
||||
onTagRenamed={handleTagRenamed}
|
||||
onTagDeleted={handleTagDeleted}
|
||||
onTagsRefresh={noop}
|
||||
humanizeTags={humanizeTags}
|
||||
/>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
// ── Stories ───────────────────────────────────────────────────────────────
|
||||
|
||||
/** Default view with a nested tag tree. All interactions are exercisable:
|
||||
* click + New / Move to open the picker modal; click a tag to filter;
|
||||
* double-click or pencil to rename; × to delete.
|
||||
* API mutations are stubbed — the tree won't re-fetch after saves,
|
||||
* but no network errors will occur. */
|
||||
export const Default = {
|
||||
render: () => <TagsViewSandbox />,
|
||||
};
|
||||
|
||||
/** Empty archive — "No tags yet." shown; + New still works. */
|
||||
export const Empty = {
|
||||
render: () => <TagsViewSandbox initialNodes={[]} />,
|
||||
};
|
||||
|
||||
/** Humanize mode: slugs displayed as Title Case names. */
|
||||
export const HumanizedNames = {
|
||||
render: () => <TagsViewSandbox humanizeTags />,
|
||||
};
|
||||
|
||||
/** Active tag filter — header shows the current filter path. */
|
||||
export const WithActiveFilter = {
|
||||
render: () => (
|
||||
<TagsViewSandbox tagFilter="/science/computer-science" />
|
||||
),
|
||||
};
|
||||
|
||||
/** Flat list with no nesting. */
|
||||
export const FlatList = {
|
||||
render: () => (
|
||||
<TagsViewSandbox
|
||||
initialNodes={[
|
||||
tag('a1', 'Books', 'books', '/books', 8),
|
||||
tag('a2', 'Films', 'films', '/films', 3),
|
||||
tag('a3', 'Music', 'music', '/music', 0),
|
||||
tag('a4', 'Podcasts', 'podcasts', '/podcasts', 14),
|
||||
]}
|
||||
/>
|
||||
),
|
||||
};
|
||||
|
||||
/** Tags with large entry counts — verifies count badge layout. */
|
||||
export const HighCounts = {
|
||||
render: () => (
|
||||
<TagsViewSandbox
|
||||
initialNodes={[
|
||||
tag('h1', 'All', 'all', '/all', 9999, [
|
||||
tag('h2', 'Starred', 'starred', '/all/starred', 432),
|
||||
tag('h3', 'Archive', 'archive', '/all/archive', 1204),
|
||||
]),
|
||||
]}
|
||||
/>
|
||||
),
|
||||
};
|
||||
|
|
@ -1,45 +0,0 @@
|
|||
import { useEffect, useState } from 'react'
|
||||
import { fetchArtifactText } from '../api'
|
||||
|
||||
/**
|
||||
* Renders the primary_media artifact of a text/document entry as plain text.
|
||||
*
|
||||
* v1 intentionally does NOT render Markdown — we don't want a Markdown parser
|
||||
* dep just to unblock the "no preview available" fallback, and monospace text
|
||||
* with visible fences reads fine for the note-length content this feature
|
||||
* captures. Bump to a real renderer if/when Markdown-authored entries grow.
|
||||
*/
|
||||
export default function TextPreview({ src, mime, title }) {
|
||||
const [text, setText] = useState(null)
|
||||
const [error, setError] = useState(null)
|
||||
|
||||
useEffect(() => {
|
||||
const controller = new AbortController()
|
||||
setText(null)
|
||||
setError(null)
|
||||
fetchArtifactText(src, { signal: controller.signal })
|
||||
.then(setText)
|
||||
.catch(e => {
|
||||
if (e.name !== 'AbortError') setError(e.message || String(e))
|
||||
})
|
||||
return () => controller.abort()
|
||||
}, [src])
|
||||
|
||||
if (error) {
|
||||
return (
|
||||
<div className="text-preview text-preview--error">
|
||||
Failed to load text: {error}
|
||||
</div>
|
||||
)
|
||||
}
|
||||
if (text === null) {
|
||||
return <div className="text-preview text-preview--loading">Loading…</div>
|
||||
}
|
||||
return (
|
||||
<div className="text-preview">
|
||||
{title && <h1 className="text-preview__title">{title}</h1>}
|
||||
<pre className="text-preview__body">{text}</pre>
|
||||
{mime && <div className="text-preview__mime">{mime}</div>}
|
||||
</div>
|
||||
)
|
||||
}
|
||||
36
frontend/src/components/Topbar.stories.jsx
Normal file
36
frontend/src/components/Topbar.stories.jsx
Normal file
|
|
@ -0,0 +1,36 @@
|
|||
import Topbar from './Topbar';
|
||||
|
||||
export default {
|
||||
component: Topbar,
|
||||
tags: ['autodocs'],
|
||||
};
|
||||
|
||||
const defaultArchives = [
|
||||
{ id: '1', label: 'Main Archive' },
|
||||
{ id: '2', label: 'Research' },
|
||||
{ id: '3', label: 'Screenshots' },
|
||||
];
|
||||
|
||||
export const Default = {
|
||||
args: {
|
||||
archives: defaultArchives,
|
||||
archiveId: '1',
|
||||
onArchiveChange: () => {},
|
||||
view: 'archive',
|
||||
onViewChange: () => {},
|
||||
onCaptureClick: () => {},
|
||||
},
|
||||
};
|
||||
|
||||
export const WithUser = {
|
||||
args: {
|
||||
...Default.args,
|
||||
},
|
||||
decorators: [
|
||||
(Story) => (
|
||||
<div style={{ background: 'var(--paper)' }}>
|
||||
<Story />
|
||||
</div>
|
||||
),
|
||||
],
|
||||
};
|
||||
102
frontend/src/components/UIPatterns.stories.jsx
Normal file
102
frontend/src/components/UIPatterns.stories.jsx
Normal file
|
|
@ -0,0 +1,102 @@
|
|||
export default {
|
||||
title: 'UI Patterns',
|
||||
tags: ['autodocs'],
|
||||
};
|
||||
|
||||
export const Buttons = () => (
|
||||
<div style={{ padding: '20px', display: 'flex', gap: '16px', flexWrap: 'wrap' }}>
|
||||
<button className="capture-button">+ Capture</button>
|
||||
<button className="nav-link">Nav Link</button>
|
||||
<button className="nav-link is-active">Nav Link (Active)</button>
|
||||
<button className="assign-tag-btn">Add Tag</button>
|
||||
<button className="assign-tag-btn" style={{ pointerEvents: 'none', opacity: 0.6 }}>Disabled</button>
|
||||
</div>
|
||||
);
|
||||
|
||||
export const FormInputs = () => (
|
||||
<div style={{ padding: '20px', maxWidth: '400px', display: 'flex', flexDirection: 'column', gap: '16px' }}>
|
||||
<div>
|
||||
<label style={{ display: 'block', marginBottom: '6px', color: 'var(--muted)', fontSize: '12px' }}>Search Input</label>
|
||||
<input className="search-input" type="search" placeholder="Search archive..." />
|
||||
</div>
|
||||
<div>
|
||||
<label style={{ display: 'block', marginBottom: '6px', color: 'var(--muted)', fontSize: '12px' }}>Capture Input</label>
|
||||
<input className="capture-input" type="text" placeholder="tweet:1234567890 or https://..." />
|
||||
</div>
|
||||
<div>
|
||||
<label style={{ display: 'block', marginBottom: '6px', color: 'var(--muted)', fontSize: '12px' }}>Tag Input</label>
|
||||
<input className="assign-tag-input" type="text" placeholder="/science/cs" />
|
||||
</div>
|
||||
<div>
|
||||
<label style={{ display: 'block', marginBottom: '6px', color: 'var(--muted)', fontSize: '12px' }}>Archive Switcher</label>
|
||||
<select className="archive-switcher" style={{ width: '100%' }}>
|
||||
<option>Main Archive</option>
|
||||
<option>Research</option>
|
||||
<option>Screenshots</option>
|
||||
</select>
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
|
||||
export const Pills = () => (
|
||||
<div style={{ padding: '20px', display: 'flex', gap: '8px', flexWrap: 'wrap' }}>
|
||||
<span className="type-pill">page</span>
|
||||
<span className="type-pill">video</span>
|
||||
<span className="type-pill">tweet_thread</span>
|
||||
<span className="type-pill">file</span>
|
||||
</div>
|
||||
);
|
||||
|
||||
export const TagPills = () => (
|
||||
<div style={{ padding: '20px', display: 'flex', gap: '8px', flexWrap: 'wrap' }}>
|
||||
<span className="tag-pill">
|
||||
science
|
||||
<button className="remove-tag">×</button>
|
||||
</span>
|
||||
<span className="tag-pill">
|
||||
computer-science
|
||||
<button className="remove-tag">×</button>
|
||||
</span>
|
||||
<span className="tag-pill">
|
||||
learning
|
||||
<button className="remove-tag">×</button>
|
||||
</span>
|
||||
</div>
|
||||
);
|
||||
|
||||
export const ColorPalette = () => (
|
||||
<div style={{ padding: '20px', display: 'grid', gridTemplateColumns: 'repeat(auto-fit, minmax(200px, 1fr))', gap: '16px' }}>
|
||||
<div>
|
||||
<div style={{ height: '80px', background: 'var(--ink)', marginBottom: '8px', borderRadius: '4px' }} />
|
||||
<strong>Ink</strong> (#20251f)
|
||||
</div>
|
||||
<div>
|
||||
<div style={{ height: '80px', background: 'var(--paper)', border: '1px solid var(--line)', marginBottom: '8px', borderRadius: '4px' }} />
|
||||
<strong>Paper</strong> (#f5f0e7)
|
||||
</div>
|
||||
<div>
|
||||
<div style={{ height: '80px', background: 'var(--accent)', marginBottom: '8px', borderRadius: '4px' }} />
|
||||
<strong>Accent</strong> (#8d3f30)
|
||||
</div>
|
||||
<div>
|
||||
<div style={{ height: '80px', background: 'var(--accent-2)', marginBottom: '8px', borderRadius: '4px' }} />
|
||||
<strong>Accent 2</strong> (#b78342)
|
||||
</div>
|
||||
<div>
|
||||
<div style={{ height: '80px', background: 'var(--link)', marginBottom: '8px', borderRadius: '4px' }} />
|
||||
<strong>Link</strong> (#245f72)
|
||||
</div>
|
||||
<div>
|
||||
<div style={{ height: '80px', background: 'var(--top)', marginBottom: '8px', borderRadius: '4px' }} />
|
||||
<strong>Top</strong> (#141d18)
|
||||
</div>
|
||||
<div>
|
||||
<div style={{ height: '80px', background: 'var(--muted)', marginBottom: '8px', borderRadius: '4px' }} />
|
||||
<strong>Muted</strong> (#666a61)
|
||||
</div>
|
||||
<div>
|
||||
<div style={{ height: '80px', background: 'var(--line)', marginBottom: '8px', borderRadius: '4px' }} />
|
||||
<strong>Line</strong> (#d2c6b5)
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
|
|
@ -302,10 +302,10 @@ select {
|
|||
border-bottom: 1px solid var(--line-soft);
|
||||
}
|
||||
#entries-body > div > div { padding: 7px 10px; flex-shrink: 0; overflow: hidden; }
|
||||
/* Pending capture rows (no index class) fall back to nth-child; real rows use explicit classes. */
|
||||
/* Skeleton rows (no index class) fall back to nth-child; real rows use explicit classes. */
|
||||
#entries-body > div:not(.entry-row-outer):nth-child(even) { background: #f2ede5; }
|
||||
#entries-body > div:not(.entry-row-outer):nth-child(odd) { background: var(--paper-3); }
|
||||
/* Index-based stripes for real entry rows — immune to pending-capture sibling count. */
|
||||
/* Index-based stripes for real entry rows — immune to skeleton sibling count. */
|
||||
#entries-body > .entry-row-outer--light { background: var(--paper-3); }
|
||||
#entries-body > .entry-row-outer--dark { background: #f2ede5; }
|
||||
#entries-body > div.is-selected {
|
||||
|
|
@ -638,9 +638,6 @@ select {
|
|||
}
|
||||
.bulk-coll-select:focus { border-color: var(--accent); }
|
||||
.tag-add-btn:disabled { opacity: 0.45; cursor: default; }
|
||||
.bulk-title-note { font-size: 12px; color: var(--muted); margin: 6px 0 0; }
|
||||
.bulk-title-note.form-msg--ok { color: var(--link); }
|
||||
.bulk-title-note.form-msg--err { color: var(--accent); }
|
||||
|
||||
/* ── Rail delete zone ───────────────────────────────────────────────────── */
|
||||
.rail-delete-zone {
|
||||
|
|
@ -955,104 +952,6 @@ select {
|
|||
line-height: 1.45;
|
||||
}
|
||||
|
||||
/* ── Text-capture row ───────────────────────────────────────────────────
|
||||
A text row breaks the [dot][one-line-input][×] shape used by URL and
|
||||
file rows: it stacks a title, a multi-line body, and a small footer
|
||||
(mime selector). Everything lives inside .capture-text-inputs so the
|
||||
outer flex-row still gives us [icon][content][×] alignment. */
|
||||
|
||||
.capture-text-row > .capture-row-main {
|
||||
/* Icon and remove button sit at the top of the block, not centered on
|
||||
the tall textarea. */
|
||||
align-items: flex-start;
|
||||
}
|
||||
|
||||
.capture-text-icon {
|
||||
flex-shrink: 0;
|
||||
width: 20px;
|
||||
height: 44px; /* matches .capture-text-title height */
|
||||
display: grid;
|
||||
place-items: center;
|
||||
color: var(--muted);
|
||||
}
|
||||
|
||||
.capture-text-inputs {
|
||||
flex: 1 1 auto;
|
||||
min-width: 0; /* let the textarea shrink inside flex */
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
gap: 8px;
|
||||
}
|
||||
|
||||
.capture-text-input {
|
||||
width: 100%;
|
||||
border: 1px solid var(--line);
|
||||
background: var(--field);
|
||||
color: var(--ink);
|
||||
border-radius: var(--r2);
|
||||
outline: none;
|
||||
transition: border-color .15s ease, box-shadow .15s ease;
|
||||
font-family: inherit;
|
||||
}
|
||||
.capture-text-input:focus {
|
||||
border-color: var(--accent);
|
||||
box-shadow: 0 0 0 3px color-mix(in srgb, var(--accent) 14%, transparent);
|
||||
}
|
||||
|
||||
.capture-text-title {
|
||||
height: 44px;
|
||||
padding: 0 14px;
|
||||
font-size: 15px;
|
||||
}
|
||||
|
||||
.capture-text-body {
|
||||
min-height: 140px;
|
||||
padding: 10px 14px;
|
||||
font-size: 14px;
|
||||
line-height: 1.55;
|
||||
resize: vertical;
|
||||
}
|
||||
.capture-text-body::placeholder,
|
||||
.capture-text-title::placeholder {
|
||||
color: var(--muted);
|
||||
}
|
||||
|
||||
.capture-text-footer {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
gap: 8px;
|
||||
justify-content: flex-end; /* mime selector sits under the body */
|
||||
}
|
||||
|
||||
/* Mime selector: reuse the small chip styling of .capture-quality. */
|
||||
.capture-text-mime {
|
||||
flex-shrink: 0;
|
||||
height: 28px;
|
||||
padding: 0 26px 0 10px;
|
||||
border: 1px solid var(--line);
|
||||
border-radius: var(--r);
|
||||
background: var(--paper);
|
||||
color: var(--ink);
|
||||
font-size: 12px;
|
||||
cursor: pointer;
|
||||
outline: none;
|
||||
transition: border-color .15s;
|
||||
/* Custom caret so it doesn't look like an unstyled OS dropdown. */
|
||||
appearance: none;
|
||||
-webkit-appearance: none;
|
||||
background-image: url("data:image/svg+xml;utf8,<svg xmlns='http://www.w3.org/2000/svg' width='12' height='12' viewBox='0 0 12 12' fill='none' stroke='%23888' stroke-width='1.75' stroke-linecap='round' stroke-linejoin='round'><path d='M3 4.5l3 3 3-3'/></svg>");
|
||||
background-repeat: no-repeat;
|
||||
background-position: right 8px center;
|
||||
}
|
||||
.capture-text-mime:hover { border-color: var(--accent-2); }
|
||||
.capture-text-mime:focus { border-color: var(--accent); }
|
||||
|
||||
/* Remove button on text row: pin to top so tall bodies don't push it. */
|
||||
.capture-text-row .capture-row-action {
|
||||
margin-top: 8px;
|
||||
}
|
||||
|
||||
|
||||
/* Add-another button */
|
||||
.capture-add-row {
|
||||
display: flex;
|
||||
|
|
@ -2100,16 +1999,6 @@ select {
|
|||
}
|
||||
.admin-table td:first-child { padding-left: 16px; }
|
||||
.admin-table tr:hover td { background: var(--paper-2); }
|
||||
/* ── Settings › yt-dlp ── */
|
||||
.ytdlp-section { max-width: 760px; }
|
||||
.ytdlp-summary { display: grid; grid-template-columns: max-content 1fr; gap: 6px 16px; margin: 0 0 12px; font-size: 14px; }
|
||||
.ytdlp-summary dt { font-weight: 700; color: var(--muted); }
|
||||
.ytdlp-summary dd { margin: 0; }
|
||||
.ytdlp-path { font-family: ui-monospace, SFMono-Regular, Menlo, monospace; font-size: 12px; word-break: break-all; }
|
||||
.ytdlp-table-wrap { overflow-x: auto; margin: 6px 0 14px; border: 1px solid var(--line-soft); }
|
||||
.ytdlp-row--chosen td { background: var(--paper-2); }
|
||||
.ytdlp-badge { font-size: 11px; font-weight: 700; color: var(--paper); background: var(--link); border-radius: 3px; padding: 1px 6px; white-space: nowrap; }
|
||||
.ytdlp-actions { display: flex; gap: 8px; align-items: center; margin: 8px 0; }
|
||||
.admin-row-disabled td { opacity: 0.45; }
|
||||
.admin-section { margin-bottom: 36px; max-width: 860px; }
|
||||
.admin-section h2 {
|
||||
|
|
@ -2947,44 +2836,36 @@ google-cast-launcher.video-tv-btn {
|
|||
/* Push content above the fixed AudioBar when it is visible */
|
||||
body.has-audio-bar { padding-bottom: 56px; }
|
||||
|
||||
/* ── In-progress entry row ──────────────────────────────────────────────── */
|
||||
.in-progress-entry-row {
|
||||
box-sizing: border-box;
|
||||
min-width: 0;
|
||||
padding: 7px 22px;
|
||||
gap: 9px;
|
||||
color: var(--muted);
|
||||
white-space: nowrap;
|
||||
}
|
||||
.in-progress-entry-row__spinner {
|
||||
width: 16px;
|
||||
height: 16px;
|
||||
border-width: 2px;
|
||||
color: var(--accent);
|
||||
flex-shrink: 0;
|
||||
}
|
||||
.in-progress-entry-row__locator {
|
||||
min-width: 0;
|
||||
overflow: hidden;
|
||||
text-overflow: ellipsis;
|
||||
font-family: ui-monospace, "SF Mono", Menlo, monospace;
|
||||
font-size: 11.5px;
|
||||
}
|
||||
.in-progress-entry-row__status {
|
||||
display: inline-flex;
|
||||
align-items: center;
|
||||
gap: 5px;
|
||||
flex-shrink: 0;
|
||||
color: var(--muted-2);
|
||||
font-size: 11.5px;
|
||||
}
|
||||
.in-progress-entry-row__kind {
|
||||
font-size: 10.5px;
|
||||
opacity: 0.85;
|
||||
/* ── Skeleton entry row ─────────────────────────────────────────────────── */
|
||||
@keyframes skeleton-shimmer {
|
||||
0% { background-position: 200% center; }
|
||||
100% { background-position: -200% center; }
|
||||
}
|
||||
|
||||
.skeleton-cell {
|
||||
display: inline-block;
|
||||
border-radius: 3px;
|
||||
background: linear-gradient(90deg, var(--paper-2) 25%, var(--line-soft) 50%, var(--paper-2) 75%);
|
||||
background-size: 200% 100%;
|
||||
animation: skeleton-shimmer 1.8s ease-in-out infinite;
|
||||
}
|
||||
|
||||
/* Row container — flex row matching #entries-body > div layout */
|
||||
.skeleton-row {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
background: var(--paper-3);
|
||||
}
|
||||
.skeleton-row > div {
|
||||
padding: 7px 10px;
|
||||
flex-shrink: 0;
|
||||
overflow: hidden;
|
||||
}
|
||||
.skeleton-row .col-added { padding-left: 22px; }
|
||||
.skeleton-row > div:last-child { padding-right: 22px; }
|
||||
|
||||
@media (pointer: coarse) {
|
||||
.in-progress-entry-row { padding-inline: 10px; }
|
||||
.skeleton-row .col-added { padding-left: 10px; }
|
||||
}
|
||||
|
||||
/* ── Child entry expansion ───────────────────────────────────────────────── */
|
||||
|
|
@ -3075,56 +2956,6 @@ body.has-audio-bar { padding-bottom: 56px; }
|
|||
.child-entry-row .col-added { padding-left: 22px; }
|
||||
.child-entry-row > div:last-child { padding-right: 22px; }
|
||||
|
||||
/* Child reorder: drag handle or move buttons inside .col-title, never both.
|
||||
The arrows are the default (touch, no pointer, phone-sized ≤640px viewports,
|
||||
and any browser that can't evaluate the query below). The one media query
|
||||
switches a mouse/trackpad on a wider viewport to the drag handle, so the two
|
||||
cases are exact complements without range syntax. Alt+↑/↓ on a focused row
|
||||
works everywhere. */
|
||||
.child-reorder-controls {
|
||||
display: inline-flex;
|
||||
align-items: center;
|
||||
gap: 4px;
|
||||
flex-shrink: 0;
|
||||
transition: opacity 0.15s;
|
||||
}
|
||||
.child-drag-handle {
|
||||
display: none;
|
||||
cursor: grab;
|
||||
padding: 0 3px;
|
||||
font-size: 0.8em;
|
||||
letter-spacing: -2px;
|
||||
user-select: none;
|
||||
color: var(--muted);
|
||||
}
|
||||
.child-drag-handle:active { cursor: grabbing; }
|
||||
.child-move-btn {
|
||||
width: 28px;
|
||||
height: 28px;
|
||||
padding: 0;
|
||||
border: none;
|
||||
background: none;
|
||||
color: inherit;
|
||||
font-size: 0.9em;
|
||||
line-height: 1;
|
||||
cursor: pointer;
|
||||
}
|
||||
.child-move-btn:disabled { opacity: 0.3; cursor: default; }
|
||||
.child-move-btn:focus { outline: none; }
|
||||
.child-move-btn:focus-visible { outline: 2px solid var(--accent); outline-offset: 1px; border-radius: 2px; }
|
||||
@media (hover: hover) and (pointer: fine) and (min-width: 641px) {
|
||||
/* Dimmed until the row is hovered/focused; there is hover to reveal it. */
|
||||
.child-reorder-controls { gap: 1px; opacity: 0.3; }
|
||||
.child-entry-row:hover .child-reorder-controls,
|
||||
.child-entry-row:focus-within .child-reorder-controls { opacity: 1; }
|
||||
.child-drag-handle { display: inline; }
|
||||
.child-move-btn { display: none; }
|
||||
}
|
||||
.child-entry-row.is-dragging { opacity: 0.4; }
|
||||
.child-entry-row.is-drop-before { box-shadow: inset 0 2px 0 var(--accent); }
|
||||
.child-entry-row.is-drop-after { box-shadow: inset 0 -2px 0 var(--accent); }
|
||||
.child-entries[aria-busy="true"] .child-entry-row { cursor: progress; }
|
||||
|
||||
/* Expand chevron button */
|
||||
.entry-expand-btn {
|
||||
display: inline-flex;
|
||||
|
|
@ -3287,178 +3118,3 @@ body.has-audio-bar { padding-bottom: 56px; }
|
|||
cursor: pointer;
|
||||
}
|
||||
.capture-sync-row input[type=checkbox] { cursor: pointer; }
|
||||
|
||||
/* ── Summary rail section ────────────────────────────────────────────────── */
|
||||
/* Reuses .rail-section spacing and .rail-rearchive-btn for the action button;
|
||||
only the summary-specific typography and the provider row are new here. */
|
||||
.rail-summary-body { margin-bottom: 10px; }
|
||||
.rail-summary-tldr {
|
||||
margin: 0 0 8px;
|
||||
font-size: 13.5px;
|
||||
font-weight: 600;
|
||||
color: var(--ink);
|
||||
line-height: 1.45;
|
||||
}
|
||||
.rail-summary-text {
|
||||
margin: 0 0 8px;
|
||||
font-size: 13px;
|
||||
color: var(--ink);
|
||||
line-height: 1.55;
|
||||
}
|
||||
.rail-summary-tags { display: flex; flex-wrap: wrap; gap: 5px; margin-bottom: 8px; }
|
||||
.rail-summary-tag {
|
||||
font-size: 11px;
|
||||
padding: 2px 7px;
|
||||
border: 1px solid var(--line);
|
||||
border-radius: 999px;
|
||||
color: var(--muted);
|
||||
}
|
||||
.rail-summary-provider {
|
||||
margin: 0;
|
||||
font-size: 11px;
|
||||
color: var(--muted-2);
|
||||
letter-spacing: 0.02em;
|
||||
}
|
||||
.rail-summary-status {
|
||||
display: flex; align-items: center; gap: 7px;
|
||||
margin: 0 0 8px;
|
||||
font-size: 12.5px;
|
||||
color: var(--muted);
|
||||
}
|
||||
.rail-summary-info {
|
||||
margin: 0 0 8px;
|
||||
padding: 8px;
|
||||
color: var(--muted);
|
||||
background: var(--paper-2);
|
||||
border: 1px solid var(--line-soft);
|
||||
border-radius: 4px;
|
||||
}
|
||||
.rail-summary-info__heading {
|
||||
margin: 0 0 4px;
|
||||
color: var(--ink);
|
||||
font-size: 12.5px;
|
||||
font-weight: 600;
|
||||
}
|
||||
.rail-summary-info__detail {
|
||||
margin: 0;
|
||||
font-size: 12px;
|
||||
line-height: 1.45;
|
||||
}
|
||||
.rail-summary-error { margin: 0 0 8px; }
|
||||
.rail-summary-spinner {
|
||||
width: 11px; height: 11px;
|
||||
border: 1.5px solid var(--line);
|
||||
border-top-color: var(--muted);
|
||||
border-radius: 50%;
|
||||
animation: rail-summary-spin 0.7s linear infinite;
|
||||
flex-shrink: 0;
|
||||
}
|
||||
@keyframes rail-summary-spin { to { transform: rotate(360deg); } }
|
||||
/* Respect a reduced-motion preference: the text alone still conveys the state. */
|
||||
@media (prefers-reduced-motion: reduce) {
|
||||
.rail-summary-spinner { animation: none; }
|
||||
}
|
||||
.rail-summary-controls { display: flex; flex-direction: column; gap: 6px; }
|
||||
.rail-summary-select {
|
||||
width: 100%;
|
||||
padding: 5px 8px;
|
||||
font-size: 12.5px;
|
||||
color: var(--ink);
|
||||
background: var(--paper);
|
||||
border: 1px solid var(--line);
|
||||
border-radius: 4px;
|
||||
}
|
||||
.rail-summary-image-option {
|
||||
padding: 7px 8px;
|
||||
color: var(--muted);
|
||||
background: var(--paper);
|
||||
border: 1px solid var(--line);
|
||||
border-radius: 4px;
|
||||
}
|
||||
.rail-summary-image-option__label {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
gap: 6px;
|
||||
color: var(--ink);
|
||||
font-size: 12.5px;
|
||||
cursor: pointer;
|
||||
}
|
||||
.rail-summary-image-option__label input[type="checkbox"] {
|
||||
margin: 0;
|
||||
accent-color: var(--accent);
|
||||
cursor: pointer;
|
||||
}
|
||||
.rail-summary-image-option__label input[type="checkbox"]:focus-visible {
|
||||
outline: 2px solid var(--accent);
|
||||
outline-offset: 2px;
|
||||
}
|
||||
.rail-summary-image-option__note {
|
||||
margin: 5px 0 0;
|
||||
font-size: 11px;
|
||||
line-height: 1.4;
|
||||
}
|
||||
.rail-summary-transcribe-note {
|
||||
margin: 0;
|
||||
font-size: 11px;
|
||||
line-height: 1.4;
|
||||
color: var(--muted);
|
||||
}
|
||||
.rail-summary-image-option--disabled {
|
||||
color: var(--muted-2);
|
||||
background: var(--paper);
|
||||
}
|
||||
.rail-summary-image-option--disabled .rail-summary-image-option__label {
|
||||
color: var(--muted-2);
|
||||
cursor: not-allowed;
|
||||
}
|
||||
.rail-summary-image-option--disabled .rail-summary-image-option__label input[type="checkbox"] {
|
||||
cursor: not-allowed;
|
||||
}
|
||||
|
||||
/* ── Text / Markdown preview ────────────────────────────────────────────── */
|
||||
.text-preview {
|
||||
padding: 32px 40px 48px;
|
||||
max-width: 780px;
|
||||
margin: 0 auto;
|
||||
font-family: var(--sans);
|
||||
color: var(--ink);
|
||||
overflow: auto;
|
||||
height: 100%;
|
||||
box-sizing: border-box;
|
||||
}
|
||||
.text-preview--loading,
|
||||
.text-preview--error {
|
||||
padding: 24px;
|
||||
color: var(--muted);
|
||||
font-size: 13px;
|
||||
}
|
||||
.text-preview--error { color: var(--accent); }
|
||||
.text-preview__title {
|
||||
font-family: var(--serif, var(--sans));
|
||||
font-size: 22px;
|
||||
font-weight: 600;
|
||||
margin: 0 0 20px;
|
||||
line-height: 1.25;
|
||||
color: var(--ink);
|
||||
}
|
||||
.text-preview__body {
|
||||
white-space: pre-wrap;
|
||||
overflow-wrap: anywhere;
|
||||
word-break: break-word;
|
||||
font-family: ui-monospace, "SF Mono", Menlo, Consolas, monospace;
|
||||
font-size: 13.5px;
|
||||
line-height: 1.65;
|
||||
color: var(--ink);
|
||||
background: var(--paper);
|
||||
border: 1px solid var(--line);
|
||||
border-radius: var(--r2);
|
||||
padding: 16px 18px;
|
||||
margin: 0;
|
||||
}
|
||||
.text-preview__mime {
|
||||
margin-top: 12px;
|
||||
font-size: 11px;
|
||||
color: var(--muted);
|
||||
letter-spacing: .04em;
|
||||
text-align: right;
|
||||
}
|
||||
|
|
|
|||
|
|
@ -128,33 +128,6 @@ in
|
|||
non-loopback address.
|
||||
'';
|
||||
};
|
||||
|
||||
environment = lib.mkOption {
|
||||
type = lib.types.attrsOf lib.types.str;
|
||||
default = { };
|
||||
example = lib.literalExpression ''
|
||||
{
|
||||
ARCHIVR_TRANSCRIBE_ENGINES = "whisper";
|
||||
ARCHIVR_WHISPER_CLI = "''${pkgs.whisper-cpp}/bin/whisper-cli";
|
||||
ARCHIVR_WHISPER_MODEL = "/var/lib/archivr-server/models/ggml-large-v3-turbo.bin";
|
||||
}
|
||||
'';
|
||||
description = ''
|
||||
Extra environment variables (e.g. ARCHIVR_TRANSCRIBE_ENGINES,
|
||||
ARCHIVR_PHONON2_CLI, LLM provider settings). Merged over defaults that
|
||||
point HOME, XDG_CACHE_HOME and HF_HOME into the state directory, so
|
||||
Python transcription engines can cache downloaded weights.
|
||||
'';
|
||||
};
|
||||
|
||||
environmentFile = lib.mkOption {
|
||||
type = lib.types.nullOr lib.types.path;
|
||||
default = null;
|
||||
description = ''
|
||||
Optional systemd EnvironmentFile (KEY=value lines) for settings that
|
||||
should stay out of the Nix store, such as LLM API keys.
|
||||
'';
|
||||
};
|
||||
};
|
||||
|
||||
config = lib.mkIf cfg.enable {
|
||||
|
|
@ -179,17 +152,10 @@ in
|
|||
wantedBy = [ "multi-user.target" ];
|
||||
after = [ "network.target" ];
|
||||
|
||||
environment = {
|
||||
HOME = lib.mkDefault "/var/lib/archivr-server";
|
||||
XDG_CACHE_HOME = lib.mkDefault "/var/lib/archivr-server/.cache";
|
||||
HF_HOME = lib.mkDefault "/var/lib/archivr-server/.cache/huggingface";
|
||||
} // cfg.environment;
|
||||
|
||||
serviceConfig = {
|
||||
ExecStart = "${cfg.package}/bin/archivr-server ${configFile}";
|
||||
User = cfg.user;
|
||||
Group = cfg.group;
|
||||
EnvironmentFile = lib.mkIf (cfg.environmentFile != null) cfg.environmentFile;
|
||||
|
||||
# State directory — auth SQLite lives here across upgrades/restarts.
|
||||
StateDirectory = "archivr-server";
|
||||
|
|
@ -203,8 +169,6 @@ in
|
|||
# Each archive_path is an .archivr dir; its sibling store/ dir (where
|
||||
# capture artifacts are written) lives at the same level. Whitelisting
|
||||
# the parent covers both without over-permissioning.
|
||||
# GPU transcription engines (CUDA) need /dev/nvidia*; adding
|
||||
# PrivateDevices or DeviceAllow here would break them.
|
||||
NoNewPrivileges = true;
|
||||
PrivateTmp = true;
|
||||
ProtectSystem = "strict";
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue