From 4090dc21752e6ae7239b6c6ad744134cafd3d851 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Wed, 2 Sep 2026 14:02:24 +0900 Subject: [PATCH 01/10] docs: add public product landing page --- docs/index.md | 50 ++++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 50 insertions(+) create mode 100644 docs/index.md diff --git a/docs/index.md b/docs/index.md new file mode 100644 index 00000000..fb93efdc --- /dev/null +++ b/docs/index.md @@ -0,0 +1,50 @@ +# Codec Carver + +[![Ask DeepWiki](https://deepwiki.com/badge.svg)](https://deepwiki.com/ContextualWisdomLab/codec-carver) + +Codec Carver turns long recordings into durable, metadata-preserving audio artifacts and provides evidence-aware tooling for organizing, transcribing, and reconciling recording libraries. + +## What it does + +- Converts supported recordings to FLAC or size-bounded Opus while preserving source metadata. +- Splits long recordings at safe duration boundaries, preferring detected silence when possible. +- Offers a Python CLI, an optional FastAPI upload surface, and an MCP integration. +- Provides a Rust-backed library-curation workflow for hashing, inventory, duplicate quarantine, TMK/VAD reconciliation, and bounded mutations. +- Supports optional transcription and evidence-backed description workflows while keeping source recordings intact. + +## Quick start + +Prerequisites are Python 3.10+ and `ffmpeg`/`ffprobe` on `PATH`. + +```bash +pip install -e . +codec-carver /path/to/recordings --execute --output-dir under_2gb +``` + +For the optional web service: + +```bash +pip install -e ".[web]" +docker build -t codec-carver . +docker run -p 8000:8000 codec-carver +``` + +See the repository README for configuration, duration splitting, metadata tagging, transcription, and the GPU/Rust library-curation workflow. + +## Architecture and operating model + +The CLI owns conversion planning and execution. The library-curation path combines Python orchestration with a Rust backend for byte-heavy scanning and mutation work. Evidence and provenance are kept explicit so later TMK or transcription information can be reconciled without silently rewriting source history. + +Architecture reference: + +- [Segmentation and reconciliation](architecture/segmentation-reconciliation.md) + +## Documentation + +Start with the [repository README](../README.md), then follow the architecture and doctoring material under `docs/` for specific operational and safety contracts. DeepWiki provides an additional navigable view of the repository: + +- [Ask DeepWiki](https://deepwiki.com/ContextualWisdomLab/codec-carver) + +## Releases and verification + +Use GitHub Releases and the repository's protected-branch history as the source of truth for shipped versions. A documentation source commit is not, by itself, evidence that a GitHub Pages deployment is live; publication should be verified from the repository's live Pages state before treating this page as a deployed site. From 14904f7412b32fb1851e909b654df21cf07b12ce Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Wed, 2 Sep 2026 14:07:56 +0900 Subject: [PATCH 02/10] docs: complete MIT source license --- LICENSE | 21 +++++++++++++++++++++ 1 file changed, 21 insertions(+) create mode 100644 LICENSE diff --git a/LICENSE b/LICENSE new file mode 100644 index 00000000..9eff520d --- /dev/null +++ b/LICENSE @@ -0,0 +1,21 @@ +MIT License + +Copyright (c) 2026 Codec Carver contributors + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. From dec36281ecea15bc345eeb58537aba06bfe11319 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Wed, 2 Sep 2026 14:08:08 +0900 Subject: [PATCH 03/10] docs: make public landing license-aware --- docs/index.md | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/docs/index.md b/docs/index.md index fb93efdc..8034a3c9 100644 --- a/docs/index.md +++ b/docs/index.md @@ -45,6 +45,10 @@ Start with the [repository README](../README.md), then follow the architecture a - [Ask DeepWiki](https://deepwiki.com/ContextualWisdomLab/codec-carver) -## Releases and verification +## Status and verification -Use GitHub Releases and the repository's protected-branch history as the source of truth for shipped versions. A documentation source commit is not, by itself, evidence that a GitHub Pages deployment is live; publication should be verified from the repository's live Pages state before treating this page as a deployed site. +The package metadata currently identifies source version `0.1.0`. Treat GitHub Releases and protected-branch history as the authority for shipped versions and release evidence; a source version or documentation commit alone is not a release. Likewise, this `docs/index.md` file is only a Pages source prerequisite until repository settings, deployment, and the live HTTPS page are independently verified. + +## License + +Codec Carver source declares the MIT license in `pyproject.toml`; this branch completes that existing source-license lineage with the root [MIT LICENSE](../LICENSE). The MIT grant applies to Codec Carver-authored source and documentation. External tools and dependencies—including `ffmpeg`/`ffprobe`, Python/Rust packages, model runtimes, model weights, and provider services—retain their own licenses and terms and are not relicensed by this repository. From b18e7fc2c84b47a6c18bd7954aff16d3ec9bfe79 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Wed, 2 Sep 2026 15:14:20 +0900 Subject: [PATCH 04/10] docs: make README product-first without losing operator detail --- README.md | 538 +++++------------------------------- docs/advanced-operations.md | 525 +++++++++++++++++++++++++++++++++++ 2 files changed, 592 insertions(+), 471 deletions(-) create mode 100644 docs/advanced-operations.md diff --git a/README.md b/README.md index 75f027cb..265dbbbf 100644 --- a/README.md +++ b/README.md @@ -1,525 +1,121 @@ # Codec Carver -Python CLI for carving long recordings into metadata-preserved FLAC/Opus files. +[![Ask DeepWiki](https://deepwiki.com/badge.svg)](https://deepwiki.com/ContextualWisdomLab/codec-carver) -For the long-recording curation contract (TMK/VAD evidence precedence, -provenance, and late-TMK selective reconciliation), see -[`docs/architecture/segmentation-reconciliation.md`](docs/architecture/segmentation-reconciliation.md). +**Turn long recordings into durable, metadata-preserving audio artifacts without losing the source.** -Convert supported audio recordings to FLAC or, only when needed to fit each output under a target size, high-bitrate Opus. The tool preserves originals and writes generated files to a separate output directory. Each generated output is kept below the configured size target and below four hours; longer sources are split at long silence intervals when possible. +Codec Carver is a Python CLI and evidence-aware recording-library toolkit. It converts supported recordings into size-bounded FLAC/Opus output, splits long audio at bounded duration points, preserves source metadata, and can optionally add transcription, library inventory, duplicate quarantine, and TMK/VAD reconciliation workflows. -## Install +The product boundary is deliberately conservative: source recordings remain authoritative, generated output is written separately, and evidence used for later library mutations stays explicit rather than being inferred from filenames or model output. -Requires Python 3.10+ and `ffmpeg`/`ffprobe` on `PATH`. +## Choose the workflow you need + +- **Convert recordings** — produce metadata-preserving FLAC/Opus artifacts under explicit size and duration limits. +- **Split long audio** — prefer detected silence before a duration boundary and fall back to a bounded hard split when necessary. +- **Add searchable transcripts** — opt into transcription sidecars without making transcription failure invalidate a completed conversion. +- **Curate a recording library** — use the Rust-backed library workflow for hashing, inventory, exact-duplicate quarantine, bounded materialization, and evidence-aware renaming/reconciliation. +- **Expose an integration surface** — optionally run the FastAPI upload service or MCP integration when those dependencies are installed. + +Codec Carver is not a media server, cloud-sync authority, or irreversible-delete tool. Provider state, authentication, access policy, model licensing, and downstream publication remain separate responsibilities. + +## Quick start + +Prerequisites are Python 3.10+ and `ffmpeg`/`ffprobe` on `PATH`. + +For a source checkout: ```bash -pip install -e . # CLI core (stdlib only) -pip install -e ".[web]" # + FastAPI upload service -pip install -e ".[mcp]" # + MCP server +pip install -e . +codec-carver /path/to/recordings --execute --output-dir under_2gb ``` -This installs the `codec-carver` console command: +Use a generated-only output directory so originals and generated artifacts remain easy to distinguish. Inspect help before a consequential batch: ```bash -codec-carver /path/to/recordings --execute --output-dir under_2gb +codec-carver --help ``` -## Web service (Docker) +For the optional web surface: ```bash +pip install -e ".[web]" docker build -t codec-carver . -docker run -p 8000:8000 codec-carver # upload UI at http://localhost:8000 +docker run -p 8000:8000 codec-carver ``` -## Verified command for this folder - -Run from `media_shrink_tool/`: +For MCP integration: ```bash -python3 media_shrinker.py .. \ - --execute \ - --download-icloud \ - --include-under-limit \ - --flac-all \ - --exclude-dir-prefix split_over \ - --max-duration-seconds 14400 \ - --workers 2 \ - --ffmpeg-threads 0 \ - --output-dir under_2gb \ - --report under_2gb/conversion_report.json +pip install -e ".[mcp]" ``` -Outputs are written under `../under_2gb/`. Existing generated output directories and `split_over*` directories should be excluded from scans to avoid reconverting generated media. Files under 2GB are included by default; use `--over-limit-only` only when intentionally processing oversized sources exclusively. +## Common usage -## Config file for repeat workflows +A repeatable conversion can be stored in `.codec-carver.json` in the scan root or current working directory. CLI options override config values, while the scan root and `--execute` remain intentionally non-configurable so a config file cannot silently turn a dry run into a mutation. -Instead of re-typing long flag sets, store them once in a `.codec-carver.json` file in the scan root (checked first) or the current working directory: +Example: ```json { - "flac_all": true, - "exclude_dir_prefix": ["split_over"], - "max_duration_seconds": 14400, - "workers": 2, - "output_dir": "under_2gb" + "flac_all": true, + "exclude_dir_prefix": ["split_over"], + "max_duration_seconds": 14400, + "workers": 2, + "output_dir": "under_2gb" } ``` -Then repeat runs collapse to `python3 media_shrinker.py .. --execute --download-icloud`. - -- Keys map 1:1 to CLI options with dashes replaced by underscores (`--target-bytes` becomes `target_bytes`). -- Explicit CLI flags always override config values; without a config file, behavior is identical to a plain invocation. -- `root` and `--execute` are intentionally not configurable: the config file is discovered via the scan root, and a config file must never silently turn a dry run into a real conversion. -- Unknown keys, wrong value types, and malformed JSON abort with a clear error listing the valid keys. -- JSON is used instead of TOML because the stdlib TOML parser requires Python 3.11+, while this project also supports Python 3.10. - -## Duration splitting +Metadata overrides such as `--set-title`, `--set-artist`, `--set-album`, and `--set-comment` are passed to `ffmpeg` as individual arguments rather than through a shell. Output formats include the default FLAC/Opus policy plus explicit FLAC, Opus, AAC, and MP3 modes where supported by the current CLI. -- `--max-duration-seconds 14400` keeps every generated file below four hours. -- When a source is at or above that duration, the tool runs FFmpeg `silencedetect` and prefers the latest safe point inside a long silence before the four-hour boundary. -- If no suitable silence is detected before a boundary, the tool hard-splits just under the configured maximum so the duration rule is still enforced. -- Split outputs are named with part suffixes, for example `meeting.wav.part0001.flac`, `meeting.wav.part0002.flac`. -- Tune silence detection with `--silence-noise` and `--silence-min-duration-seconds` when recordings need stricter or looser silence boundaries. +## Recording-library curation -## Metadata tagging +The importable library API is `audio_library.AudioLibrary`; the CLI entry point is `codec-carver-library`. -- `--set-title`, `--set-artist`, `--set-album`, and `--set-comment` stamp the corresponding tags on every generated output, so archived files stay searchable in players and music libraries. -- Generated commands already copy source metadata with `-map_metadata 0`; the `--set-*` values are injected after it, so each provided key overrides that specific source tag while all other source metadata is preserved (standard ffmpeg semantics). -- When none of the `--set-*` options are passed, generated ffmpeg commands are byte-identical to the untagged behavior. -- Values are passed to ffmpeg as single argv items without a shell, so spaces, quotes, and other special characters are safe as given. +The library path combines Python orchestration with a Rust backend for byte-heavy inventory and mutation work. It uses SHA-256-bound evidence, keeps exact-duplicate quarantine recoverable, and records the provenance needed to reconcile later TMK or transcript information without silently rewriting source history. -```bash -python3 media_shrinker.py .. --execute \ - --set-album "Board Meetings 2026" \ - --set-comment "archived by codec-carver" -``` +Apple Silicon/MLX, CUDA transcription, iCloud/File Provider staging, TMK hydration, description review, backend pinning, and low-disk operating procedures are intentionally kept out of the customer landing page. They remain available in the preserved [advanced operations reference](docs/advanced-operations.md) and the architecture documentation. -## Output format +## Architecture and safety boundary -- `--format auto` (default) keeps the original behaviour: FLAC for lossless (or `--flac-all`) input, high-bitrate Opus otherwise. -- `--format flac` / `--format opus` force that codec. -- `--format aac` (`.m4a`) and `--format mp3` produce broadly-compatible lossy output fitted to the target size — useful for players/devices that don't handle FLAC or Opus. +The major responsibilities are: -## Transcription (optional) +1. **Conversion CLI** — scans selected input, plans conversion/splitting, and writes generated media separately from source. +2. **Library orchestration** — manages inventory, transcript evidence, plans, and recoverable mutation state. +3. **Rust backend** — performs byte-heavy scanning, hashing, bounded filesystem work, and mutation primitives used by the library path. +4. **External tools/models** — `ffmpeg`/`ffprobe`, Python/Rust packages, model runtimes, model weights, and provider services remain independent dependencies with their own authority and licensing. -Turn each shrunk recording into searchable text. With `--transcribe`, a text and -JSON transcript sidecar is written next to every generated audio file -(`recording.wav.flac` → `recording.wav.flac.txt` / `.json`): +Start with [segmentation and reconciliation](docs/architecture/segmentation-reconciliation.md) for the TMK/VAD evidence contract. The deeper GPU/Rust design is documented in [the library architecture](docs/architecture/gpu-transcription-rust-backend.md). -```bash -python3 media_shrinker.py .. --execute --output-dir under_2gb --transcribe -``` +## Verification and status -Transcription is opt-in and uses [`faster-whisper`](https://github.com/SYSTRAN/faster-whisper), -imported lazily. Install it to enable the feature: +Run the repository test surface from the source tree: ```bash -pip install faster-whisper # then pass --transcribe +python3 -m unittest discover -s tests +python3 -m py_compile media_shrinker.py ``` -If it is not installed, conversion runs normally and transcription is skipped -with a `TRANSCRIBE_SKIP` notice. A failing transcript never aborts a conversion. -Choose a model with `--transcribe-model` (default `base`). +The package metadata currently identifies source version `0.1.0`. Treat that as source metadata, not by itself as proof of a published release, supported deployment, benchmark, customer adoption, or certification. GitHub Releases and exact protected-branch evidence are the authority for shipped artifacts when such artifacts exist. -## GPU audio-library curation (Python API + Rust backend) +Current source also contains CI, fuzzing, SAST, and security workflows; use the results for the exact revision you intend to ship rather than predecessor-head evidence. -The audio-library workflow standardizes recording names from recording time, -known location, transcript content, and SHA-256; parses Sony `.tmk` markers; and -quarantines exact duplicates. Byte-heavy scanning and mutations run in Rust, -while Python keeps one GPU transcription model loaded for the batch. The -default MLX path jointly transcribes and separates anonymous speakers with -MOSS; legacy Whisper remains available explicitly. Ollama is never used and GPU -mode does not fall back to CPU. +## Documentation -The editable install below is for local checkout development only. The hardened -persistent macOS GPU bootstrap installs hash-locked dependencies and runs the -checkout directly instead of installing the project editable. +- [Documentation home](docs/index.md) +- [Advanced operations reference](docs/advanced-operations.md) +- [Segmentation and reconciliation](docs/architecture/segmentation-reconciliation.md) +- [GPU transcription / Rust backend architecture](docs/architecture/gpu-transcription-rust-backend.md) +- [Security policy](SECURITY.md) +- [Ask DeepWiki](https://deepwiki.com/ContextualWisdomLab/codec-carver) -```bash -cargo build --release --manifest-path rust-core/Cargo.toml -python3.12 -m venv .venv -.venv/bin/pip install -e ".[transcribe-mlx,describe-mlx]" # Apple Silicon / Metal - -codec-carver-library /path/to/recordings inventory --threads 4 -# Refresh only already-known paths after Finder materializes them. Rust hashes -# exactly these files and Python atomically merges them into the full manifest, -# avoiding unrelated multi-gigabyte iCloud reads. -codec-carver-library /path/to/recordings inventory \ - --path 'FOLDER01/231102_1840(1).wav' \ - --path 'FOLDER01/231102_1840(1).tmk' -# When the recording root is in iCloud, keep mutable evidence state on local -# storage so File Provider cannot roll back an inventory or mutation journal. -codec-carver-library /path/to/recordings \ - --state-dir "$HOME/Library/Application Support/codec-carver/sony-icd-tx650" \ - inventory --path 'FOLDER01/231102_1840(1).wav' -# Queue only explicitly selected dataless files through native FileManager and -# return immediately. Repeat --path for a deliberately bounded download batch. -codec-carver-library /path/to/recordings materialize \ - --path 'FOLDER01/231113_1524.wav' \ - --path 'FOLDER01/231113_1524(1).wav' -codec-carver-library /path/to/recordings hydrate-tmk --workers 4 -codec-carver-library /path/to/recordings hydrate-tmk \ - --workers 1 --path 'FOLDER01/231101_0917.tmk' -codec-carver-library /path/to/recordings stream-transcribe --accelerator mlx -# If a TMK arrives after a fixed-range fallback, bind its verified SHA and get -# a promote-or-selective-reprocess plan without deleting the old transcript. -codec-carver-library /path/to/recordings reconcile-tmk \ - --path 'FOLDER01/recording.wav' -# Speaker-aware MLX transcription is the default. Each SHA-keyed .txt contains -# one dialogue file with consecutive turns rendered as `[S01] ...`, `[S02] ...`. -# The pinned 0.9B MOSS model transcribes Korean and assigns timestamps and -# anonymous speakers in one Metal pass; Ollama and CPU transcription are unused. -# For a deliberately bounded small batch, pipeline iCloud reads in Rust with -# ordered, single-model GPU transcription. -codec-carver-library /path/to/recordings stream-transcribe --accelerator mlx \ - --prefetch-workers 4 --prefetch-max-bytes 536870912 -# Use legacy Whisper explicitly when word-level audit evidence is required. -codec-carver-library /path/to/recordings stream-transcribe --accelerator mlx \ - --no-speaker-diarization --model mlx-community/whisper-large-v3-turbo-q4 \ - --word-timestamps -# Summarize verified transcripts into filename topics with pinned Gemma 4 on -# Metal. This calls MLX-VLM directly; no Ollama server or transcript upload is -# involved. Repeat --path to keep the description batch bounded. -codec-carver-library /path/to/recordings describe \ - --path "recording-a.m4a" --path "recording-b.wav" -# Bind a reviewer-corrected central-context title to exact one-based MLX -# word-timestamp segments. Repeat --segment-id for direct supporting passages. -codec-carver-library /path/to/recordings review-description \ - --path "recording-b.wav" \ - --title "VOC건수보다-정보질이중요하고-활용공유하며-등록절차가간소화" \ - --central-idea "VOC 포상은 건수 최다 등록자가 합니다. 정보 질이 많이 떨어진 것 같습니다. 활용을 투명하게 공유하고 공감을 많이 받은 정보에 혜택을 연결하고 등록 절차를 간소화해야 합니다." \ - --outcome "활용을 투명하게 공유하고 공감을 많이 받은 정보에 혜택을 연결하고 등록 절차를 간소화해야 합니다." \ - --segment-id 164 --segment-id 263 --segment-id 317 --segment-id 318 \ - --segment-id 359 --segment-id 362 --segment-id 444 --segment-id 467 \ - --segment-id 891 --confidence high -codec-carver-library /path/to/recordings plan -# Bound both planning and later apply-time revalidation to one audio record and -# its linked TMK. Repeat --path for an explicitly selected batch. -codec-carver-library /path/to/recordings plan \ - --path "FOLDER01/231018_1018.wav" -# Every name is compared with the complete SHA-bound name derived from its -# transcript and drift is reported. Changing an existing standard name requires -# one of these explicit refresh authorizations. -codec-carver-library /path/to/recordings plan \ - --refresh-standardized-path "2024-06-24_15-44-11__선유로__old-title__sha256-04d93e2e12fb.m4a" -codec-carver-library /path/to/recordings plan \ - --refresh-description-drift --defer-unready -# When iCloud has not supplied every source, mutate only fully ready recordings -# and preserve the unresolved paths as explicit deferred evidence. -codec-carver-library /path/to/recordings plan --defer-unready -codec-carver-library /path/to/recordings apply # validation only -codec-carver-library /path/to/recordings apply --execute -``` +## Contributing and support -The library backend is loaded only from the repository's release/debug build or -an explicit `--backend-binary` accompanied by `--backend-sha256`; it is never -selected from ambient `PATH`. The selected binary must be owner-controlled, -non-symlinked, and non-group/world-writable. Python copies the exact bytes read -from a stable, no-follow source descriptor into an independent owner-only -execution inode, seals its directory, and forces every Rust command to that -SHA-256-pinned snapshot. Replacing the configured source path after validation -therefore cannot change the bytes that execute. Duration probing uses only the -approved fixed system `ffprobe` locations; ambient environment variables cannot -change the selected executable. Rust, ffprobe, and ffmpeg children all -receive a minimal allowlisted environment that excludes `LD_*` and `DYLD_*` -loader injection controls. MLX-VLM preflight additionally uses Python isolated -mode, a trusted runtime working directory, and verifies the package origin is -beneath that interpreter's prefix before importing native model code. -The approved absolute `ffmpeg` decodes MLX audio before it is passed to MOSS or -Whisper as an in-memory waveform, so the model libraries never resolve a bare -`ffmpeg` from caller-controlled `PATH`. Transcription repositories are also -immutable inputs: MLX Whisper accepts only -`mlx-community/whisper-large-v3-turbo-q4` at revision -`660c343bbf4e52ac257f0b7d952e5388e6f93bef`, while CUDA resolves -`dropbox-dash/faster-whisper-large-v3-turbo` at revision -`0a363e9161cbc7ed1431c9597a8ceaf0c4f78fcf`. Mutable model names or arbitrary -Hub repositories are rejected before inference. Speaker-aware MLX accepts only -`OpenMOSS-Team/MOSS-Transcribe-Diarize` at revision -`e8681d68e7042738ffca8ac8212bc8fcb1131ab8`. - -`describe` loads the pinned 4-bit -`mlx-community/gemma-4-e2b-it-4bit` revision once per batch, samples up to 48 -GPU transcript segments across the full recording, and first extracts one central idea, -outcome, confidence level, and cited segment IDs. A separate title pass must -express that context instead of listing frequent keywords; low-confidence or -generic-only titles are deferred. The final title and its audit context are -cached together in the SHA-keyed transcript sidecar, and evidence selection is -rescored against both the thesis and outcome. The model identifier and revision -are allowlisted, tokenizer remote code is disabled, transcript prompt data is -control-delimiter escaped JSON, and every title term must be recoverable from -the transcript itself. Segment references count only when they appear as -anchored `[S###]` labels; an `S###` string inside speech is not evidence. -Untrusted CR/LF and other control whitespace inside each Whisper segment are -collapsed before Python assigns its label, and the resulting labels must form -the exact contiguous sequence `S001`, `S002`, and so on. Title grounding -preserves token boundaries, so a cross-token substring cannot impersonate a -source term. Central idea and outcome terms must also occur in the cited -transcript segments, so the model cannot legitimize an invented title through -its own analysis fields. When a speaker explicitly marks a conclusion with -phrases such as `결론`, `종합하면`, or `하고 싶은 말`, at least one such segment -must support the analysis. The same conclusion IDs and their neighboring -context survive every repair prompt; if the small model still fails, a literal -fallback may compose a title only from those exact conclusion clauses and then -run the full grounding checks again. A dense explicit directive may use two -directly related evidence segments without padding a long recording with an -unrelated third segment; it still runs the same literal grounding checks. Old -keyword-only caches are not silently upgraded. Planning consumes this -evidence-backed description when present and retains the deterministic extractor -as a no-model failure-safe. -`review-description` provides the corresponding bounded correction path for a -reviewer who has inspected the full transcript. It accepts only a SHA-verified -MLX transcript with word timestamps or joint speaker-segment timestamps and a -pinned transcription revision, copies the exact selected segment text and time -ranges into an owner-only evidence record, and validates the central idea, -outcome, and title against those passages before replacing an automatic title. -The review never edits raw -transcript text. Review-time compound clauses may add Korean grammatical -particles only when at least three transcript-derived semantic terms remain in -the same filename token. Incomplete connective clauses and pronoun-only -observations are rejected as non-outcomes. The selected original segment IDs, -derived evidence IDs, transcription model/revision, and review timestamp remain -auditable in the SHA-keyed sidecar and `manual-description-review.json`. -Once semantic analysis has explicitly failed, its reason is checkpointed and -the unstandardized recording is deferred instead of being renamed from a -keyword-only fallback. Planning reports an existing standard name when its -entire basename differs from the timestamp, location, transcript-derived -central-context title, and SHA suffix recomputed from current evidence, but a -durable rename still requires explicit refresh authorization. -An evidence-backed title that cannot fit the macOS NFD UTF-8 filename budget is -rejected instead of being silently cut into a different or incomplete claim; -the reviewer must approve a shorter title whose complete meaning fits. - -`materialize` is the nonblocking iCloud request mode. Rust validates each -explicit audio/TMK path beneath the library root, rejects symlinks, calls -Foundation's `startDownloadingUbiquitousItem` only for a dataless placeholder, -and reports whether the request was queued or the file was already local. -Python rechecks the current dataless flag, updates the inventory, and writes an -owner-only `materialization-run.json`; it does not infer that accepted requests -have finished. This keeps download selection bounded while Finder is locked or -unavailable. - -`stream-transcribe` is the low-disk iCloud mode: by default Rust streams one -remote file to system scratch while calculating SHA-256, Metal/CUDA transcribes -that local stage, and Python atomically checkpoints before removing the stage. -The default selection order keeps already-materialized recordings ahead of -remote placeholders for throughput. Add `--oldest-first` when lineage work must -select the globally earliest `recorded_at` across nested directories before -local availability. When timestamps tie, an original-looking path is selected -before numbered copy suffixes; the run checkpoint records the chosen order. -Already-materialized recordings follow the same byte-binding rule: Rust opens -each path component with no-follow descriptors, copies and hashes the opened -file into private scratch, and the GPU reads only that verified copy. A pathname -swap after inspection therefore cannot redirect transcription outside the -library. Python independently opens the backend-reported scratch child relative -to its owner-only directory with `O_NOFOLLOW` and requires the scratch file to -have exactly one link. It confirms that name still identifies the opened inode, -unlinks the name, and only then hashes the anonymous descriptor. The actual byte -count and SHA-256 must match both the backend record and any known inventory -digest before ffmpeg or faster-whisper consumes that same descriptor. A -same-user hardlink, replacement path, or post-check rename therefore cannot -redirect the bytes used for inference. -`--prefetch-workers` keeps a bounded rolling queue of Rust/iCloud staging calls -full; `--prefetch-max-bytes` caps their combined logical size (512 MiB by -default). As soon as the next selected recording is staged, the ordered Python -loop starts its single-model GPU transcription while later Rust staging futures -continue in the same bounded pool. GPU work, durable checkpoints, scratch -removal, and native eviction remain serialized. The run summary records the -number of GPU calls that actually overlapped unfinished prefetch work as -`prefetch_transcription_overlaps`. Native eviction is deferred while any bounded -stage is still running so it cannot contend with FileProvider prefetch. The -no-progress stage timeout defaults to 420 seconds because real iCloud -placeholders can take more than two minutes to -deliver their first byte; override it with `--stage-stall-timeout-seconds` when -the provider has a different latency envelope. A parallel prefetch that reaches -that timeout is retried once through the serial staging path, after bounded -parallel stages finish, because FileProvider can defer every concurrent request -while accepting an immediate single request. If that serial canary also fails, -later timeouts in the same batch skip the otherwise identical long retry; other -failures are not retried. The run summary records fallback attempts, recoveries, -and suppressions. A terminal native-stage stall is checkpointed as -`error_code: stage_source_stalled` with `timeout_seconds`, -`stage_progress_bytes`, and `retryable: true`; its readable error points to an -unhealthy iCloud/FileProvider materialization path instead of exposing only a -generic subprocess command timeout. Already local files stay local. -Run `hydrate-tmk` -first when iCloud holds Sony sidecars: -it reads the tiny TMK files concurrently, checkpoints each SHA-256 and the full -ordered marker vector, and backfills any existing transcript sidecars. The -transcript provenance stores the verified primary `tmk_sha256` alongside its -path and marker vector; unresolved or stale TMK identity is recorded as null. -When File Provider status is unknown, only small TMK sidecars use the bounded -direct-read probe; long audio remains on the coordinated, checkpointed path. -Verified -TMK offsets split long MLX recordings into bounded, one-second-overlap decode -ranges while the same pinned Whisper model remains resident; midpoint ownership -removes overlap duplicates and restores every segment to its recording-global -timestamp. This avoids decoding an hours-long recording into one peak-memory -waveform. When a recording longer than ten minutes has no usable TMK vector, -MLX falls back to deterministic five-minute ranges with the same overlap, -global timestamps, and per-range checkpointing. The transcript distinguishes -`tmk_markers`, `fixed_duration`, and `single_pass` chunking, and the run summary -counts newly completed `automatic_chunked_recordings`. A later dataless flag -does not cause the same TMK to be downloaded again. Four workers and a 60-second -per-file timeout are the defaults because higher iCloud File Provider concurrency -can delay every placeholder; rerunning resumes only unresolved sidecars. Repeat -`--path` to verify only the TMKs paired with the bounded audio batch instead of -waking every iCloud placeholder. Already verified TMKs also repair stale linked -transcript metadata without rehashing; `synced_transcripts` and `sync_failed` -report that idempotent pass separately from new TMK hydration. -`stream-transcribe` never blocks an audio recording on an unresolved TMK: it uses -hydrated markers when present and records `tmk_error` evidence otherwise. -If that primary sidecar is still remote but a same-directory, same-time, -same-size TMK with an equivalent copy-normalized stem has a content-verified SHA -and valid ordered markers, streaming may use it only as a bounded decode hint. -The transcript keeps the unresolved primary `tmk_path` and separately records -the hint path, SHA-256, marker count, last marker, and full vector; it never -presents the sibling as the primary sidecar. `tmk_chunk_hints_used` reports this -performance fallback per run. -Gemma title generation also keeps its two-to-six-token quality gate. If a final -literal-evidence repair still exceeds that bound, codec-carver deterministically -rebuilds a subject-purpose title only from the already validated central idea, -outcome, cited transcript evidence, and transcript-grounded terms instead of -accepting or blindly truncating the model output. -Inventory validation also requires every audio `tmk_path` to reference a record -whose kind is exactly `tmk`; a crafted audio-to-audio link cannot authorize -quarantining canonical audio as if it were a duplicate sidecar. -On macOS, Rust requests every dataless item through Foundation's supported -`FileManager.startDownloadingUbiquitousItem` API, then coordinates the read with -`NSFileCoordinator` and performs the single-pass copy-and-hash inside the -coordinated accessor. The coordinator is required by current File Provider -domains to keep `isDownloadRequested`/`isDownloading` active; already-local -files keep the direct fast path. The implementation does not depend on the -undocumented `brctl download` command. If Finder and the coordinated native -request both remain at zero bytes, inspect File Provider with -`fileproviderctl check` before an operator-approved repair. -After a durable transcript checkpoint, Rust also releases the local source -blocks through `FileManager.evictUbiquitousItem`; no `brctl evict` subprocess is -used. Eviction is optional cleanup, so a native eviction error is recorded in -`eviction_failures` without converting a completed transcription into a failure. -At startup it samples the live macOS dataless flag and drains currently local -audio before remote placeholders, keeping the GPU fed while iCloud catches up. -Rust stage monitoring resets its stall clock whenever the partial grows; the -default 420-second stall limit skips only placeholders making no byte progress, -not large files that are actively copying and hashing. An independent absolute -deadline, four times the configured stall limit, also bounds repeated premature -EOF retries even when a faulty provider reports monotonically increasing byte -counts. File Provider can expose -the logical source size before any bytes are readable; Rust rejects such a -premature short/empty EOF, and Python retries it only until the same bounded -zero-progress deadline instead of accepting the empty-file SHA-256. -Batch commands still print their complete JSON checkpoint summary, but return a -non-zero process status when any selected file is recorded in `failures`. -Planning rejects recordings without SHA-256 or transcript evidence by default. -`--defer-unready` keeps those paths unchanged and lists them in -`deferred_paths`, allowing verified subsets to proceed without inventing a -placeholder description. `plan --path` narrows quarantine and rename operations -to the selected audio paths and their linked TMKs; the same selection is stored -in the private plan and recomputed at apply time, while omitting it preserves the -whole-library batch behavior. -Every rescan archives the previous inventory by its SHA-256. If iCloud evicts a -previously hashed recording, same-path/same-size evidence and transcript -sidecars restore its full hash only as an explicitly unverified identity hint. -It cannot form an exact-duplicate group or a new rename/quarantine operation -until Rust hashes current bytes. Audio and TMK duplicate groups are tracked -separately, so a same-SHA TMK sidecar never collides with an audio record. An -executed mutation journal can restore -identity continuity after a move, but remains unverified until current bytes are -opened and hashed again. Materialized files are rehashed before any transcript -cache hit or new mutation plan, then copied and hashed into private scratch -before a GPU call. -Transcripts are keyed by the full SHA-256 under -`.codec-carver/transcripts/`, use owner-only directory/file permissions, and -accept only canonical 64-hex digest filenames. Every transcript consumer opens -the final sidecar relative to a verified directory descriptor with -`O_NOFOLLOW`; symlinks and non-regular sidecars are unavailable evidence, never -external JSON input. Cache, planning, TMK backfill, and inventory reconciliation -also verify the sidecar's embedded SHA-256 against its inventory record; a -foreign sidecar cannot suppress GPU inference or supply a filename title. Exact -copies are inferred only once. Ultra-short -low-confidence words remain auditable in JSON but do not enter standardized -filenames. For long meetings, the optional Gemma phase records the central idea, -outcome, confidence, and directly supporting segment IDs before it creates the -filename title. Generic keyword bundles are rejected, while the deterministic -corpus-central phrase remains the no-model failure-safe. A structurally valid -timestamp/location/SHA wrapper cannot hide an arbitrary description: the -complete expected name is compared and listed in `description_drift_paths`, -while explicit refresh authorization controls the durable rename. -Duplicate files move to the recoverable -`.codec-carver/quarantine/exact-duplicates/` tree; no irreversible deletion is -performed by default. Inventory, TMK, transcript, and mutation paths are -validated beneath the canonical library root at both the public Python bridge -and Rust boundary. Direct `inspect`, `stage`, and `evict` calls reject absolute, -parent, non-portable, and symlink-component paths before launching Rust. -Symlinked state/staging roots are refused, and scratch cleanup uses a -no-follow directory handle rather than a check-then-unlink pathname. -Private state paths are created and opened from `/` one component at a time with -`mkdirat`/`openat`, `O_DIRECTORY`, and `O_NOFOLLOW`; an intermediate ancestor -swap cannot redirect an atomic state write outside the selected library. -Rust holds an exclusive per-library mutation lock from validation through -execution, walks or creates every source/destination parent relative to the -locked root descriptor with `O_NOFOLLOW`, and performs no-overwrite -descriptor-relative renames (`RENAME_EXCL` on macOS, `RENAME_NOREPLACE` on -Linux). Rollback uses the same primitive, so replacing a destination parent -with a symlink cannot redirect a move outside the library. Python refuses -`apply --execute` for injected or substitute backends; only the concrete, -descriptor-safe `RustBackend` may cross the mutation boundary. -Rust returns inventory and mutation-journal JSON on stdout; Python alone commits -those state files through descriptor-relative atomic replacement. Final-name -symlinks are never followed, and a partial or schema-invalid mutation journal is -moved to `.codec-carver/recovery/malformed-journals/` so a damaged checkpoint -cannot brick later inventories. Both recovery path components are created and -opened from the verified state-directory descriptor with `mkdirat`/`openat` -semantics, so an intermediate symlink cannot redirect quarantine outside the -library. - -The importable API is `audio_library.AudioLibrary`. The architecture, evidence -precedence, filename contract, and primary research/standards sources are in -[`docs/architecture/gpu-transcription-rust-backend.md`](docs/architecture/gpu-transcription-rust-backend.md). - -### Persistent macOS GPU runtime - -On macOS, do not place the MLX environment in an iCloud/File Provider-backed -repository. Loading native packages such as `tokenizers`, `torch`, and -`mlx-vlm` can otherwise block inside `dyld` even when the package files appear -materialized. Create the persistent runtime under the local cache instead. The -bootstrap supports Apple Silicon and installs the complete Python dependency -graph from `requirements-macos-mlx-lock.txt` with package hashes verified; the -checkout itself is run directly rather than installed as an editable package. -The script resets `PATH` before its first helper call, uses fixed system-tool -paths, and copies the reviewed SHA-256-pinned `uv` executable into the validated -runtime inode before executing it. A different reviewed `uv` build requires -both `--uv-bin` and its `--uv-sha256` digest. -The runtime must be a direct child of the owner-controlled -`~/Library/Caches/codec-carver/venvs` directory; bootstrap operations stay bound -to the validated directory inode so a later pathname swap cannot redirect them: +Keep product-facing changes evidence-bound and preserve the separation between source recordings, generated output, library evidence, and external provider/model authority. Changes to filesystem mutation, model provenance, or source identity should include focused regression tests and update the owning architecture document rather than expanding the README into an operator runbook again. -```bash -./scripts/bootstrap_macos_gpu_runtime.sh -GPU_PY="$HOME/Library/Caches/codec-carver/venvs/gpu-py312/bin/python" -"$GPU_PY" "$PWD/audio_library.py" /path/to/library inventory -"$GPU_PY" "$PWD/audio_library.py" /path/to/library transcribe --accelerator mlx -"$GPU_PY" "$PWD/audio_library.py" /path/to/library describe -``` +For suspected vulnerabilities, follow `SECURITY.md` and avoid public disclosure before coordinated review. -The bootstrap installs the hash-locked dependency sets for `transcribe-mlx` and -`describe-mlx` into one reusable environment outside File Provider storage. The Python API -keeps the Whisper and Gemma models resident for batch work, Apple Metal performs -the model inference without Ollama or CPU fallback, and the Rust backend retains -streaming SHA-256, TMK parsing, inventory, and mutation work. +## License -## Safety notes +Codec Carver source declares the **MIT License** in `pyproject.toml`; this documentation branch carries the matching root [LICENSE](LICENSE). The MIT grant applies to Codec Carver-authored source and documentation. -- Source files selected by the scan are protected from deletion or overwrite; keep `--output-dir` as a generated-only directory so excluded originals are never mistaken for stale generated outputs. -- Generated output names include the original filename and suffix, for example `clip.wav.flac` and `clip.m4a.flac`, so same-stem inputs cannot collide during parallel conversion. -- For lossy sources, `--flac-all` first creates FLAC to avoid additional loss; if that output exceeds the target size, the generated FLAC is removed and a high-bitrate Opus output is created instead. -- Filesystem metadata preservation is best effort: permissions, nanosecond access/modified times, extended attributes, and macOS creation date are copied when the operating system allows it. -- Video-containing files with supported container extensions are rejected unless - `--allow-video` is set to extract their audio track. -- For real media runs, keep `--output-dir` as a generated-only directory such as `under_2gb` and avoid `--overwrite` unless that directory contains no original source files. - -## Verification - -```bash -python3 -m unittest discover -s tests -python3 -m py_compile media_shrinker.py -``` +External tools and dependencies—including `ffmpeg`/`ffprobe`, Python/Rust packages, model runtimes, model weights, and provider services—retain their own licenses and terms and are not relicensed by this repository. A permissive repository license is not evidence that every optional model, binary, or service is approved for a particular commercial distribution. diff --git a/docs/advanced-operations.md b/docs/advanced-operations.md new file mode 100644 index 00000000..75f027cb --- /dev/null +++ b/docs/advanced-operations.md @@ -0,0 +1,525 @@ +# Codec Carver + +Python CLI for carving long recordings into metadata-preserved FLAC/Opus files. + +For the long-recording curation contract (TMK/VAD evidence precedence, +provenance, and late-TMK selective reconciliation), see +[`docs/architecture/segmentation-reconciliation.md`](docs/architecture/segmentation-reconciliation.md). + +Convert supported audio recordings to FLAC or, only when needed to fit each output under a target size, high-bitrate Opus. The tool preserves originals and writes generated files to a separate output directory. Each generated output is kept below the configured size target and below four hours; longer sources are split at long silence intervals when possible. + +## Install + +Requires Python 3.10+ and `ffmpeg`/`ffprobe` on `PATH`. + +```bash +pip install -e . # CLI core (stdlib only) +pip install -e ".[web]" # + FastAPI upload service +pip install -e ".[mcp]" # + MCP server +``` + +This installs the `codec-carver` console command: + +```bash +codec-carver /path/to/recordings --execute --output-dir under_2gb +``` + +## Web service (Docker) + +```bash +docker build -t codec-carver . +docker run -p 8000:8000 codec-carver # upload UI at http://localhost:8000 +``` + +## Verified command for this folder + +Run from `media_shrink_tool/`: + +```bash +python3 media_shrinker.py .. \ + --execute \ + --download-icloud \ + --include-under-limit \ + --flac-all \ + --exclude-dir-prefix split_over \ + --max-duration-seconds 14400 \ + --workers 2 \ + --ffmpeg-threads 0 \ + --output-dir under_2gb \ + --report under_2gb/conversion_report.json +``` + +Outputs are written under `../under_2gb/`. Existing generated output directories and `split_over*` directories should be excluded from scans to avoid reconverting generated media. Files under 2GB are included by default; use `--over-limit-only` only when intentionally processing oversized sources exclusively. + +## Config file for repeat workflows + +Instead of re-typing long flag sets, store them once in a `.codec-carver.json` file in the scan root (checked first) or the current working directory: + +```json +{ + "flac_all": true, + "exclude_dir_prefix": ["split_over"], + "max_duration_seconds": 14400, + "workers": 2, + "output_dir": "under_2gb" +} +``` + +Then repeat runs collapse to `python3 media_shrinker.py .. --execute --download-icloud`. + +- Keys map 1:1 to CLI options with dashes replaced by underscores (`--target-bytes` becomes `target_bytes`). +- Explicit CLI flags always override config values; without a config file, behavior is identical to a plain invocation. +- `root` and `--execute` are intentionally not configurable: the config file is discovered via the scan root, and a config file must never silently turn a dry run into a real conversion. +- Unknown keys, wrong value types, and malformed JSON abort with a clear error listing the valid keys. +- JSON is used instead of TOML because the stdlib TOML parser requires Python 3.11+, while this project also supports Python 3.10. + +## Duration splitting + +- `--max-duration-seconds 14400` keeps every generated file below four hours. +- When a source is at or above that duration, the tool runs FFmpeg `silencedetect` and prefers the latest safe point inside a long silence before the four-hour boundary. +- If no suitable silence is detected before a boundary, the tool hard-splits just under the configured maximum so the duration rule is still enforced. +- Split outputs are named with part suffixes, for example `meeting.wav.part0001.flac`, `meeting.wav.part0002.flac`. +- Tune silence detection with `--silence-noise` and `--silence-min-duration-seconds` when recordings need stricter or looser silence boundaries. + +## Metadata tagging + +- `--set-title`, `--set-artist`, `--set-album`, and `--set-comment` stamp the corresponding tags on every generated output, so archived files stay searchable in players and music libraries. +- Generated commands already copy source metadata with `-map_metadata 0`; the `--set-*` values are injected after it, so each provided key overrides that specific source tag while all other source metadata is preserved (standard ffmpeg semantics). +- When none of the `--set-*` options are passed, generated ffmpeg commands are byte-identical to the untagged behavior. +- Values are passed to ffmpeg as single argv items without a shell, so spaces, quotes, and other special characters are safe as given. + +```bash +python3 media_shrinker.py .. --execute \ + --set-album "Board Meetings 2026" \ + --set-comment "archived by codec-carver" +``` + +## Output format + +- `--format auto` (default) keeps the original behaviour: FLAC for lossless (or `--flac-all`) input, high-bitrate Opus otherwise. +- `--format flac` / `--format opus` force that codec. +- `--format aac` (`.m4a`) and `--format mp3` produce broadly-compatible lossy output fitted to the target size — useful for players/devices that don't handle FLAC or Opus. + +## Transcription (optional) + +Turn each shrunk recording into searchable text. With `--transcribe`, a text and +JSON transcript sidecar is written next to every generated audio file +(`recording.wav.flac` → `recording.wav.flac.txt` / `.json`): + +```bash +python3 media_shrinker.py .. --execute --output-dir under_2gb --transcribe +``` + +Transcription is opt-in and uses [`faster-whisper`](https://github.com/SYSTRAN/faster-whisper), +imported lazily. Install it to enable the feature: + +```bash +pip install faster-whisper # then pass --transcribe +``` + +If it is not installed, conversion runs normally and transcription is skipped +with a `TRANSCRIBE_SKIP` notice. A failing transcript never aborts a conversion. +Choose a model with `--transcribe-model` (default `base`). + +## GPU audio-library curation (Python API + Rust backend) + +The audio-library workflow standardizes recording names from recording time, +known location, transcript content, and SHA-256; parses Sony `.tmk` markers; and +quarantines exact duplicates. Byte-heavy scanning and mutations run in Rust, +while Python keeps one GPU transcription model loaded for the batch. The +default MLX path jointly transcribes and separates anonymous speakers with +MOSS; legacy Whisper remains available explicitly. Ollama is never used and GPU +mode does not fall back to CPU. + +The editable install below is for local checkout development only. The hardened +persistent macOS GPU bootstrap installs hash-locked dependencies and runs the +checkout directly instead of installing the project editable. + +```bash +cargo build --release --manifest-path rust-core/Cargo.toml +python3.12 -m venv .venv +.venv/bin/pip install -e ".[transcribe-mlx,describe-mlx]" # Apple Silicon / Metal + +codec-carver-library /path/to/recordings inventory --threads 4 +# Refresh only already-known paths after Finder materializes them. Rust hashes +# exactly these files and Python atomically merges them into the full manifest, +# avoiding unrelated multi-gigabyte iCloud reads. +codec-carver-library /path/to/recordings inventory \ + --path 'FOLDER01/231102_1840(1).wav' \ + --path 'FOLDER01/231102_1840(1).tmk' +# When the recording root is in iCloud, keep mutable evidence state on local +# storage so File Provider cannot roll back an inventory or mutation journal. +codec-carver-library /path/to/recordings \ + --state-dir "$HOME/Library/Application Support/codec-carver/sony-icd-tx650" \ + inventory --path 'FOLDER01/231102_1840(1).wav' +# Queue only explicitly selected dataless files through native FileManager and +# return immediately. Repeat --path for a deliberately bounded download batch. +codec-carver-library /path/to/recordings materialize \ + --path 'FOLDER01/231113_1524.wav' \ + --path 'FOLDER01/231113_1524(1).wav' +codec-carver-library /path/to/recordings hydrate-tmk --workers 4 +codec-carver-library /path/to/recordings hydrate-tmk \ + --workers 1 --path 'FOLDER01/231101_0917.tmk' +codec-carver-library /path/to/recordings stream-transcribe --accelerator mlx +# If a TMK arrives after a fixed-range fallback, bind its verified SHA and get +# a promote-or-selective-reprocess plan without deleting the old transcript. +codec-carver-library /path/to/recordings reconcile-tmk \ + --path 'FOLDER01/recording.wav' +# Speaker-aware MLX transcription is the default. Each SHA-keyed .txt contains +# one dialogue file with consecutive turns rendered as `[S01] ...`, `[S02] ...`. +# The pinned 0.9B MOSS model transcribes Korean and assigns timestamps and +# anonymous speakers in one Metal pass; Ollama and CPU transcription are unused. +# For a deliberately bounded small batch, pipeline iCloud reads in Rust with +# ordered, single-model GPU transcription. +codec-carver-library /path/to/recordings stream-transcribe --accelerator mlx \ + --prefetch-workers 4 --prefetch-max-bytes 536870912 +# Use legacy Whisper explicitly when word-level audit evidence is required. +codec-carver-library /path/to/recordings stream-transcribe --accelerator mlx \ + --no-speaker-diarization --model mlx-community/whisper-large-v3-turbo-q4 \ + --word-timestamps +# Summarize verified transcripts into filename topics with pinned Gemma 4 on +# Metal. This calls MLX-VLM directly; no Ollama server or transcript upload is +# involved. Repeat --path to keep the description batch bounded. +codec-carver-library /path/to/recordings describe \ + --path "recording-a.m4a" --path "recording-b.wav" +# Bind a reviewer-corrected central-context title to exact one-based MLX +# word-timestamp segments. Repeat --segment-id for direct supporting passages. +codec-carver-library /path/to/recordings review-description \ + --path "recording-b.wav" \ + --title "VOC건수보다-정보질이중요하고-활용공유하며-등록절차가간소화" \ + --central-idea "VOC 포상은 건수 최다 등록자가 합니다. 정보 질이 많이 떨어진 것 같습니다. 활용을 투명하게 공유하고 공감을 많이 받은 정보에 혜택을 연결하고 등록 절차를 간소화해야 합니다." \ + --outcome "활용을 투명하게 공유하고 공감을 많이 받은 정보에 혜택을 연결하고 등록 절차를 간소화해야 합니다." \ + --segment-id 164 --segment-id 263 --segment-id 317 --segment-id 318 \ + --segment-id 359 --segment-id 362 --segment-id 444 --segment-id 467 \ + --segment-id 891 --confidence high +codec-carver-library /path/to/recordings plan +# Bound both planning and later apply-time revalidation to one audio record and +# its linked TMK. Repeat --path for an explicitly selected batch. +codec-carver-library /path/to/recordings plan \ + --path "FOLDER01/231018_1018.wav" +# Every name is compared with the complete SHA-bound name derived from its +# transcript and drift is reported. Changing an existing standard name requires +# one of these explicit refresh authorizations. +codec-carver-library /path/to/recordings plan \ + --refresh-standardized-path "2024-06-24_15-44-11__선유로__old-title__sha256-04d93e2e12fb.m4a" +codec-carver-library /path/to/recordings plan \ + --refresh-description-drift --defer-unready +# When iCloud has not supplied every source, mutate only fully ready recordings +# and preserve the unresolved paths as explicit deferred evidence. +codec-carver-library /path/to/recordings plan --defer-unready +codec-carver-library /path/to/recordings apply # validation only +codec-carver-library /path/to/recordings apply --execute +``` + +The library backend is loaded only from the repository's release/debug build or +an explicit `--backend-binary` accompanied by `--backend-sha256`; it is never +selected from ambient `PATH`. The selected binary must be owner-controlled, +non-symlinked, and non-group/world-writable. Python copies the exact bytes read +from a stable, no-follow source descriptor into an independent owner-only +execution inode, seals its directory, and forces every Rust command to that +SHA-256-pinned snapshot. Replacing the configured source path after validation +therefore cannot change the bytes that execute. Duration probing uses only the +approved fixed system `ffprobe` locations; ambient environment variables cannot +change the selected executable. Rust, ffprobe, and ffmpeg children all +receive a minimal allowlisted environment that excludes `LD_*` and `DYLD_*` +loader injection controls. MLX-VLM preflight additionally uses Python isolated +mode, a trusted runtime working directory, and verifies the package origin is +beneath that interpreter's prefix before importing native model code. +The approved absolute `ffmpeg` decodes MLX audio before it is passed to MOSS or +Whisper as an in-memory waveform, so the model libraries never resolve a bare +`ffmpeg` from caller-controlled `PATH`. Transcription repositories are also +immutable inputs: MLX Whisper accepts only +`mlx-community/whisper-large-v3-turbo-q4` at revision +`660c343bbf4e52ac257f0b7d952e5388e6f93bef`, while CUDA resolves +`dropbox-dash/faster-whisper-large-v3-turbo` at revision +`0a363e9161cbc7ed1431c9597a8ceaf0c4f78fcf`. Mutable model names or arbitrary +Hub repositories are rejected before inference. Speaker-aware MLX accepts only +`OpenMOSS-Team/MOSS-Transcribe-Diarize` at revision +`e8681d68e7042738ffca8ac8212bc8fcb1131ab8`. + +`describe` loads the pinned 4-bit +`mlx-community/gemma-4-e2b-it-4bit` revision once per batch, samples up to 48 +GPU transcript segments across the full recording, and first extracts one central idea, +outcome, confidence level, and cited segment IDs. A separate title pass must +express that context instead of listing frequent keywords; low-confidence or +generic-only titles are deferred. The final title and its audit context are +cached together in the SHA-keyed transcript sidecar, and evidence selection is +rescored against both the thesis and outcome. The model identifier and revision +are allowlisted, tokenizer remote code is disabled, transcript prompt data is +control-delimiter escaped JSON, and every title term must be recoverable from +the transcript itself. Segment references count only when they appear as +anchored `[S###]` labels; an `S###` string inside speech is not evidence. +Untrusted CR/LF and other control whitespace inside each Whisper segment are +collapsed before Python assigns its label, and the resulting labels must form +the exact contiguous sequence `S001`, `S002`, and so on. Title grounding +preserves token boundaries, so a cross-token substring cannot impersonate a +source term. Central idea and outcome terms must also occur in the cited +transcript segments, so the model cannot legitimize an invented title through +its own analysis fields. When a speaker explicitly marks a conclusion with +phrases such as `결론`, `종합하면`, or `하고 싶은 말`, at least one such segment +must support the analysis. The same conclusion IDs and their neighboring +context survive every repair prompt; if the small model still fails, a literal +fallback may compose a title only from those exact conclusion clauses and then +run the full grounding checks again. A dense explicit directive may use two +directly related evidence segments without padding a long recording with an +unrelated third segment; it still runs the same literal grounding checks. Old +keyword-only caches are not silently upgraded. Planning consumes this +evidence-backed description when present and retains the deterministic extractor +as a no-model failure-safe. +`review-description` provides the corresponding bounded correction path for a +reviewer who has inspected the full transcript. It accepts only a SHA-verified +MLX transcript with word timestamps or joint speaker-segment timestamps and a +pinned transcription revision, copies the exact selected segment text and time +ranges into an owner-only evidence record, and validates the central idea, +outcome, and title against those passages before replacing an automatic title. +The review never edits raw +transcript text. Review-time compound clauses may add Korean grammatical +particles only when at least three transcript-derived semantic terms remain in +the same filename token. Incomplete connective clauses and pronoun-only +observations are rejected as non-outcomes. The selected original segment IDs, +derived evidence IDs, transcription model/revision, and review timestamp remain +auditable in the SHA-keyed sidecar and `manual-description-review.json`. +Once semantic analysis has explicitly failed, its reason is checkpointed and +the unstandardized recording is deferred instead of being renamed from a +keyword-only fallback. Planning reports an existing standard name when its +entire basename differs from the timestamp, location, transcript-derived +central-context title, and SHA suffix recomputed from current evidence, but a +durable rename still requires explicit refresh authorization. +An evidence-backed title that cannot fit the macOS NFD UTF-8 filename budget is +rejected instead of being silently cut into a different or incomplete claim; +the reviewer must approve a shorter title whose complete meaning fits. + +`materialize` is the nonblocking iCloud request mode. Rust validates each +explicit audio/TMK path beneath the library root, rejects symlinks, calls +Foundation's `startDownloadingUbiquitousItem` only for a dataless placeholder, +and reports whether the request was queued or the file was already local. +Python rechecks the current dataless flag, updates the inventory, and writes an +owner-only `materialization-run.json`; it does not infer that accepted requests +have finished. This keeps download selection bounded while Finder is locked or +unavailable. + +`stream-transcribe` is the low-disk iCloud mode: by default Rust streams one +remote file to system scratch while calculating SHA-256, Metal/CUDA transcribes +that local stage, and Python atomically checkpoints before removing the stage. +The default selection order keeps already-materialized recordings ahead of +remote placeholders for throughput. Add `--oldest-first` when lineage work must +select the globally earliest `recorded_at` across nested directories before +local availability. When timestamps tie, an original-looking path is selected +before numbered copy suffixes; the run checkpoint records the chosen order. +Already-materialized recordings follow the same byte-binding rule: Rust opens +each path component with no-follow descriptors, copies and hashes the opened +file into private scratch, and the GPU reads only that verified copy. A pathname +swap after inspection therefore cannot redirect transcription outside the +library. Python independently opens the backend-reported scratch child relative +to its owner-only directory with `O_NOFOLLOW` and requires the scratch file to +have exactly one link. It confirms that name still identifies the opened inode, +unlinks the name, and only then hashes the anonymous descriptor. The actual byte +count and SHA-256 must match both the backend record and any known inventory +digest before ffmpeg or faster-whisper consumes that same descriptor. A +same-user hardlink, replacement path, or post-check rename therefore cannot +redirect the bytes used for inference. +`--prefetch-workers` keeps a bounded rolling queue of Rust/iCloud staging calls +full; `--prefetch-max-bytes` caps their combined logical size (512 MiB by +default). As soon as the next selected recording is staged, the ordered Python +loop starts its single-model GPU transcription while later Rust staging futures +continue in the same bounded pool. GPU work, durable checkpoints, scratch +removal, and native eviction remain serialized. The run summary records the +number of GPU calls that actually overlapped unfinished prefetch work as +`prefetch_transcription_overlaps`. Native eviction is deferred while any bounded +stage is still running so it cannot contend with FileProvider prefetch. The +no-progress stage timeout defaults to 420 seconds because real iCloud +placeholders can take more than two minutes to +deliver their first byte; override it with `--stage-stall-timeout-seconds` when +the provider has a different latency envelope. A parallel prefetch that reaches +that timeout is retried once through the serial staging path, after bounded +parallel stages finish, because FileProvider can defer every concurrent request +while accepting an immediate single request. If that serial canary also fails, +later timeouts in the same batch skip the otherwise identical long retry; other +failures are not retried. The run summary records fallback attempts, recoveries, +and suppressions. A terminal native-stage stall is checkpointed as +`error_code: stage_source_stalled` with `timeout_seconds`, +`stage_progress_bytes`, and `retryable: true`; its readable error points to an +unhealthy iCloud/FileProvider materialization path instead of exposing only a +generic subprocess command timeout. Already local files stay local. +Run `hydrate-tmk` +first when iCloud holds Sony sidecars: +it reads the tiny TMK files concurrently, checkpoints each SHA-256 and the full +ordered marker vector, and backfills any existing transcript sidecars. The +transcript provenance stores the verified primary `tmk_sha256` alongside its +path and marker vector; unresolved or stale TMK identity is recorded as null. +When File Provider status is unknown, only small TMK sidecars use the bounded +direct-read probe; long audio remains on the coordinated, checkpointed path. +Verified +TMK offsets split long MLX recordings into bounded, one-second-overlap decode +ranges while the same pinned Whisper model remains resident; midpoint ownership +removes overlap duplicates and restores every segment to its recording-global +timestamp. This avoids decoding an hours-long recording into one peak-memory +waveform. When a recording longer than ten minutes has no usable TMK vector, +MLX falls back to deterministic five-minute ranges with the same overlap, +global timestamps, and per-range checkpointing. The transcript distinguishes +`tmk_markers`, `fixed_duration`, and `single_pass` chunking, and the run summary +counts newly completed `automatic_chunked_recordings`. A later dataless flag +does not cause the same TMK to be downloaded again. Four workers and a 60-second +per-file timeout are the defaults because higher iCloud File Provider concurrency +can delay every placeholder; rerunning resumes only unresolved sidecars. Repeat +`--path` to verify only the TMKs paired with the bounded audio batch instead of +waking every iCloud placeholder. Already verified TMKs also repair stale linked +transcript metadata without rehashing; `synced_transcripts` and `sync_failed` +report that idempotent pass separately from new TMK hydration. +`stream-transcribe` never blocks an audio recording on an unresolved TMK: it uses +hydrated markers when present and records `tmk_error` evidence otherwise. +If that primary sidecar is still remote but a same-directory, same-time, +same-size TMK with an equivalent copy-normalized stem has a content-verified SHA +and valid ordered markers, streaming may use it only as a bounded decode hint. +The transcript keeps the unresolved primary `tmk_path` and separately records +the hint path, SHA-256, marker count, last marker, and full vector; it never +presents the sibling as the primary sidecar. `tmk_chunk_hints_used` reports this +performance fallback per run. +Gemma title generation also keeps its two-to-six-token quality gate. If a final +literal-evidence repair still exceeds that bound, codec-carver deterministically +rebuilds a subject-purpose title only from the already validated central idea, +outcome, cited transcript evidence, and transcript-grounded terms instead of +accepting or blindly truncating the model output. +Inventory validation also requires every audio `tmk_path` to reference a record +whose kind is exactly `tmk`; a crafted audio-to-audio link cannot authorize +quarantining canonical audio as if it were a duplicate sidecar. +On macOS, Rust requests every dataless item through Foundation's supported +`FileManager.startDownloadingUbiquitousItem` API, then coordinates the read with +`NSFileCoordinator` and performs the single-pass copy-and-hash inside the +coordinated accessor. The coordinator is required by current File Provider +domains to keep `isDownloadRequested`/`isDownloading` active; already-local +files keep the direct fast path. The implementation does not depend on the +undocumented `brctl download` command. If Finder and the coordinated native +request both remain at zero bytes, inspect File Provider with +`fileproviderctl check` before an operator-approved repair. +After a durable transcript checkpoint, Rust also releases the local source +blocks through `FileManager.evictUbiquitousItem`; no `brctl evict` subprocess is +used. Eviction is optional cleanup, so a native eviction error is recorded in +`eviction_failures` without converting a completed transcription into a failure. +At startup it samples the live macOS dataless flag and drains currently local +audio before remote placeholders, keeping the GPU fed while iCloud catches up. +Rust stage monitoring resets its stall clock whenever the partial grows; the +default 420-second stall limit skips only placeholders making no byte progress, +not large files that are actively copying and hashing. An independent absolute +deadline, four times the configured stall limit, also bounds repeated premature +EOF retries even when a faulty provider reports monotonically increasing byte +counts. File Provider can expose +the logical source size before any bytes are readable; Rust rejects such a +premature short/empty EOF, and Python retries it only until the same bounded +zero-progress deadline instead of accepting the empty-file SHA-256. +Batch commands still print their complete JSON checkpoint summary, but return a +non-zero process status when any selected file is recorded in `failures`. +Planning rejects recordings without SHA-256 or transcript evidence by default. +`--defer-unready` keeps those paths unchanged and lists them in +`deferred_paths`, allowing verified subsets to proceed without inventing a +placeholder description. `plan --path` narrows quarantine and rename operations +to the selected audio paths and their linked TMKs; the same selection is stored +in the private plan and recomputed at apply time, while omitting it preserves the +whole-library batch behavior. +Every rescan archives the previous inventory by its SHA-256. If iCloud evicts a +previously hashed recording, same-path/same-size evidence and transcript +sidecars restore its full hash only as an explicitly unverified identity hint. +It cannot form an exact-duplicate group or a new rename/quarantine operation +until Rust hashes current bytes. Audio and TMK duplicate groups are tracked +separately, so a same-SHA TMK sidecar never collides with an audio record. An +executed mutation journal can restore +identity continuity after a move, but remains unverified until current bytes are +opened and hashed again. Materialized files are rehashed before any transcript +cache hit or new mutation plan, then copied and hashed into private scratch +before a GPU call. +Transcripts are keyed by the full SHA-256 under +`.codec-carver/transcripts/`, use owner-only directory/file permissions, and +accept only canonical 64-hex digest filenames. Every transcript consumer opens +the final sidecar relative to a verified directory descriptor with +`O_NOFOLLOW`; symlinks and non-regular sidecars are unavailable evidence, never +external JSON input. Cache, planning, TMK backfill, and inventory reconciliation +also verify the sidecar's embedded SHA-256 against its inventory record; a +foreign sidecar cannot suppress GPU inference or supply a filename title. Exact +copies are inferred only once. Ultra-short +low-confidence words remain auditable in JSON but do not enter standardized +filenames. For long meetings, the optional Gemma phase records the central idea, +outcome, confidence, and directly supporting segment IDs before it creates the +filename title. Generic keyword bundles are rejected, while the deterministic +corpus-central phrase remains the no-model failure-safe. A structurally valid +timestamp/location/SHA wrapper cannot hide an arbitrary description: the +complete expected name is compared and listed in `description_drift_paths`, +while explicit refresh authorization controls the durable rename. +Duplicate files move to the recoverable +`.codec-carver/quarantine/exact-duplicates/` tree; no irreversible deletion is +performed by default. Inventory, TMK, transcript, and mutation paths are +validated beneath the canonical library root at both the public Python bridge +and Rust boundary. Direct `inspect`, `stage`, and `evict` calls reject absolute, +parent, non-portable, and symlink-component paths before launching Rust. +Symlinked state/staging roots are refused, and scratch cleanup uses a +no-follow directory handle rather than a check-then-unlink pathname. +Private state paths are created and opened from `/` one component at a time with +`mkdirat`/`openat`, `O_DIRECTORY`, and `O_NOFOLLOW`; an intermediate ancestor +swap cannot redirect an atomic state write outside the selected library. +Rust holds an exclusive per-library mutation lock from validation through +execution, walks or creates every source/destination parent relative to the +locked root descriptor with `O_NOFOLLOW`, and performs no-overwrite +descriptor-relative renames (`RENAME_EXCL` on macOS, `RENAME_NOREPLACE` on +Linux). Rollback uses the same primitive, so replacing a destination parent +with a symlink cannot redirect a move outside the library. Python refuses +`apply --execute` for injected or substitute backends; only the concrete, +descriptor-safe `RustBackend` may cross the mutation boundary. +Rust returns inventory and mutation-journal JSON on stdout; Python alone commits +those state files through descriptor-relative atomic replacement. Final-name +symlinks are never followed, and a partial or schema-invalid mutation journal is +moved to `.codec-carver/recovery/malformed-journals/` so a damaged checkpoint +cannot brick later inventories. Both recovery path components are created and +opened from the verified state-directory descriptor with `mkdirat`/`openat` +semantics, so an intermediate symlink cannot redirect quarantine outside the +library. + +The importable API is `audio_library.AudioLibrary`. The architecture, evidence +precedence, filename contract, and primary research/standards sources are in +[`docs/architecture/gpu-transcription-rust-backend.md`](docs/architecture/gpu-transcription-rust-backend.md). + +### Persistent macOS GPU runtime + +On macOS, do not place the MLX environment in an iCloud/File Provider-backed +repository. Loading native packages such as `tokenizers`, `torch`, and +`mlx-vlm` can otherwise block inside `dyld` even when the package files appear +materialized. Create the persistent runtime under the local cache instead. The +bootstrap supports Apple Silicon and installs the complete Python dependency +graph from `requirements-macos-mlx-lock.txt` with package hashes verified; the +checkout itself is run directly rather than installed as an editable package. +The script resets `PATH` before its first helper call, uses fixed system-tool +paths, and copies the reviewed SHA-256-pinned `uv` executable into the validated +runtime inode before executing it. A different reviewed `uv` build requires +both `--uv-bin` and its `--uv-sha256` digest. +The runtime must be a direct child of the owner-controlled +`~/Library/Caches/codec-carver/venvs` directory; bootstrap operations stay bound +to the validated directory inode so a later pathname swap cannot redirect them: + +```bash +./scripts/bootstrap_macos_gpu_runtime.sh +GPU_PY="$HOME/Library/Caches/codec-carver/venvs/gpu-py312/bin/python" +"$GPU_PY" "$PWD/audio_library.py" /path/to/library inventory +"$GPU_PY" "$PWD/audio_library.py" /path/to/library transcribe --accelerator mlx +"$GPU_PY" "$PWD/audio_library.py" /path/to/library describe +``` + +The bootstrap installs the hash-locked dependency sets for `transcribe-mlx` and +`describe-mlx` into one reusable environment outside File Provider storage. The Python API +keeps the Whisper and Gemma models resident for batch work, Apple Metal performs +the model inference without Ollama or CPU fallback, and the Rust backend retains +streaming SHA-256, TMK parsing, inventory, and mutation work. + +## Safety notes + +- Source files selected by the scan are protected from deletion or overwrite; keep `--output-dir` as a generated-only directory so excluded originals are never mistaken for stale generated outputs. +- Generated output names include the original filename and suffix, for example `clip.wav.flac` and `clip.m4a.flac`, so same-stem inputs cannot collide during parallel conversion. +- For lossy sources, `--flac-all` first creates FLAC to avoid additional loss; if that output exceeds the target size, the generated FLAC is removed and a high-bitrate Opus output is created instead. +- Filesystem metadata preservation is best effort: permissions, nanosecond access/modified times, extended attributes, and macOS creation date are copied when the operating system allows it. +- Video-containing files with supported container extensions are rejected unless + `--allow-video` is set to extract their audio track. +- For real media runs, keep `--output-dir` as a generated-only directory such as `under_2gb` and avoid `--overwrite` unless that directory contains no original source files. + +## Verification + +```bash +python3 -m unittest discover -s tests +python3 -m py_compile media_shrinker.py +``` From bfbb72dd58f1702d6d9108782fcff23ae974ff21 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Wed, 2 Sep 2026 15:14:48 +0900 Subject: [PATCH 05/10] docs: point operator detail to advanced reference --- docs/index.md | 9 ++++++--- 1 file changed, 6 insertions(+), 3 deletions(-) diff --git a/docs/index.md b/docs/index.md index 8034a3c9..26d97e68 100644 --- a/docs/index.md +++ b/docs/index.md @@ -29,7 +29,7 @@ docker build -t codec-carver . docker run -p 8000:8000 codec-carver ``` -See the repository README for configuration, duration splitting, metadata tagging, transcription, and the GPU/Rust library-curation workflow. +Use the repository README for the product overview and common workflow. Detailed configuration, duration-splitting controls, metadata tagging, transcription, iCloud/TMK handling, and GPU/Rust library-curation procedures are preserved in the [advanced operations reference](advanced-operations.md). ## Architecture and operating model @@ -38,13 +38,16 @@ The CLI owns conversion planning and execution. The library-curation path combin Architecture reference: - [Segmentation and reconciliation](architecture/segmentation-reconciliation.md) +- [GPU transcription / Rust backend](architecture/gpu-transcription-rust-backend.md) ## Documentation -Start with the [repository README](../README.md), then follow the architecture and doctoring material under `docs/` for specific operational and safety contracts. DeepWiki provides an additional navigable view of the repository: - +- [Repository README](../README.md) +- [Advanced operations reference](advanced-operations.md) - [Ask DeepWiki](https://deepwiki.com/ContextualWisdomLab/codec-carver) +Follow the architecture and doctoring material under `docs/` for specific operational and safety contracts. + ## Status and verification The package metadata currently identifies source version `0.1.0`. Treat GitHub Releases and protected-branch history as the authority for shipped versions and release evidence; a source version or documentation commit alone is not a release. Likewise, this `docs/index.md` file is only a Pages source prerequisite until repository settings, deployment, and the live HTTPS page are independently verified. From 8851c7cf7b02fde3187e477858936469e7165223 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Wed, 2 Sep 2026 16:26:23 +0900 Subject: [PATCH 06/10] docs: make Pages root navigation durable --- docs/index.md | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/docs/index.md b/docs/index.md index 26d97e68..2490b3d0 100644 --- a/docs/index.md +++ b/docs/index.md @@ -42,7 +42,7 @@ Architecture reference: ## Documentation -- [Repository README](../README.md) +- [Repository README](https://github.com/ContextualWisdomLab/codec-carver/blob/main/README.md) - [Advanced operations reference](advanced-operations.md) - [Ask DeepWiki](https://deepwiki.com/ContextualWisdomLab/codec-carver) @@ -54,4 +54,4 @@ The package metadata currently identifies source version `0.1.0`. Treat GitHub R ## License -Codec Carver source declares the MIT license in `pyproject.toml`; this branch completes that existing source-license lineage with the root [MIT LICENSE](../LICENSE). The MIT grant applies to Codec Carver-authored source and documentation. External tools and dependencies—including `ffmpeg`/`ffprobe`, Python/Rust packages, model runtimes, model weights, and provider services—retain their own licenses and terms and are not relicensed by this repository. +Codec Carver source declares the MIT license in `pyproject.toml`; this branch completes that existing source-license lineage with the root [MIT LICENSE](https://github.com/ContextualWisdomLab/codec-carver/blob/main/LICENSE). The MIT grant applies to Codec Carver-authored source and documentation. External tools and dependencies—including `ffmpeg`/`ffprobe`, Python/Rust packages, model runtimes, model weights, and provider services—retain their own licenses and terms and are not relicensed by this repository. From eceab1623b4b0e1899a0fa8cc95029f6d1547a46 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Wed, 2 Sep 2026 17:12:23 +0900 Subject: [PATCH 07/10] docs: fix relocated architecture links --- docs/advanced-operations.md | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/docs/advanced-operations.md b/docs/advanced-operations.md index 75f027cb..baeefb30 100644 --- a/docs/advanced-operations.md +++ b/docs/advanced-operations.md @@ -4,7 +4,7 @@ Python CLI for carving long recordings into metadata-preserved FLAC/Opus files. For the long-recording curation contract (TMK/VAD evidence precedence, provenance, and late-TMK selective reconciliation), see -[`docs/architecture/segmentation-reconciliation.md`](docs/architecture/segmentation-reconciliation.md). +[`docs/architecture/segmentation-reconciliation.md`](architecture/segmentation-reconciliation.md). Convert supported audio recordings to FLAC or, only when needed to fit each output under a target size, high-bitrate Opus. The tool preserves originals and writes generated files to a separate output directory. Each generated output is kept below the configured size target and below four hours; longer sources are split at long silence intervals when possible. @@ -474,7 +474,7 @@ library. The importable API is `audio_library.AudioLibrary`. The architecture, evidence precedence, filename contract, and primary research/standards sources are in -[`docs/architecture/gpu-transcription-rust-backend.md`](docs/architecture/gpu-transcription-rust-backend.md). +[`docs/architecture/gpu-transcription-rust-backend.md`](architecture/gpu-transcription-rust-backend.md). ### Persistent macOS GPU runtime From 1b718117c4f3567e866b2cee564c05fb66a3cbb6 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Wed, 2 Sep 2026 19:01:03 +0900 Subject: [PATCH 08/10] docs: carry package license artifact metadata --- pyproject.toml | 1 + 1 file changed, 1 insertion(+) diff --git a/pyproject.toml b/pyproject.toml index 91884dfe..000e450f 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -75,6 +75,7 @@ codec-carver = "media_shrinker:main" codec-carver-library = "audio_library:main" [tool.setuptools] +license-files = ["LICENSE"] py-modules = [ "chapters", "config_file", From 8f4fc67004cd834928b342b77a9697003b81ab18 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Wed, 2 Sep 2026 19:01:23 +0900 Subject: [PATCH 09/10] docs: preserve commercial runtime boundary --- docs/index.md | 9 ++++++++- 1 file changed, 8 insertions(+), 1 deletion(-) diff --git a/docs/index.md b/docs/index.md index 2490b3d0..04797fb6 100644 --- a/docs/index.md +++ b/docs/index.md @@ -44,6 +44,7 @@ Architecture reference: - [Repository README](https://github.com/ContextualWisdomLab/codec-carver/blob/main/README.md) - [Advanced operations reference](advanced-operations.md) +- [GitHub Releases](https://github.com/ContextualWisdomLab/codec-carver/releases) - [Ask DeepWiki](https://deepwiki.com/ContextualWisdomLab/codec-carver) Follow the architecture and doctoring material under `docs/` for specific operational and safety contracts. @@ -52,6 +53,12 @@ Follow the architecture and doctoring material under `docs/` for specific operat The package metadata currently identifies source version `0.1.0`. Treat GitHub Releases and protected-branch history as the authority for shipped versions and release evidence; a source version or documentation commit alone is not a release. Likewise, this `docs/index.md` file is only a Pages source prerequisite until repository settings, deployment, and the live HTTPS page are independently verified. +## Commercial runtime boundary + +Codec Carver-authored source is MIT-licensed, but the current conversion/probing implementation requires FFmpeg/FFprobe. FFmpeg builds can carry LGPL/GPL-family obligations that are outside ContextualWisdomLab's supported commercial inbound baseline. Issue [#513](https://github.com/ContextualWisdomLab/codec-carver/issues/513) owns replacement of that execution boundary. Until that replacement is integrated and released, do not present the current FFmpeg-backed conversion path as a commercially approved deployment, and do not treat process or container separation as a license exception. + +Optional packages, native runtimes, models, weights, and provider services retain their own licenses and require profile-specific approval. + ## License -Codec Carver source declares the MIT license in `pyproject.toml`; this branch completes that existing source-license lineage with the root [MIT LICENSE](https://github.com/ContextualWisdomLab/codec-carver/blob/main/LICENSE). The MIT grant applies to Codec Carver-authored source and documentation. External tools and dependencies—including `ffmpeg`/`ffprobe`, Python/Rust packages, model runtimes, model weights, and provider services—retain their own licenses and terms and are not relicensed by this repository. +Codec Carver source declares the MIT license in `pyproject.toml`; this branch completes that existing source-license lineage with the root [MIT LICENSE](https://github.com/ContextualWisdomLab/codec-carver/blob/main/LICENSE) and includes the license file in setuptools package artifacts. The MIT grant applies to Codec Carver-authored source and documentation. External tools and dependencies—including `ffmpeg`/`ffprobe`, Python/Rust packages, model runtimes, model weights, and provider services—retain their own licenses and terms and are not relicensed by this repository. From f42925ad049e30c24b81b4035a9e85fc5c3f944d Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Tue, 8 Sep 2026 17:19:02 +0900 Subject: [PATCH 10/10] fix: restore Atheris coverage compatibility --- .github/workflows/fuzz.yml | 2 +- CHANGELOG.md | 1 + CLAUDE.md | 2 +- fuzz/README.md | 3 ++- fuzz/fuzz_build_segments.py | 2 +- fuzz/fuzz_parse_probe_payload.py | 2 +- fuzz/fuzz_parse_silencedetect.py | 2 +- fuzz/requirements-fuzz.txt | 9 ++++----- tests/test_ci_workflow.py | 21 ++++++++++++++++++++- 9 files changed, 32 insertions(+), 12 deletions(-) diff --git a/.github/workflows/fuzz.yml b/.github/workflows/fuzz.yml index c506cf7f..aaf7890a 100644 --- a/.github/workflows/fuzz.yml +++ b/.github/workflows/fuzz.yml @@ -46,7 +46,7 @@ jobs: persist-credentials: false - uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0 with: - # Atheris supports CPython 3.6 - 3.12. + # The pinned Atheris artifacts support CPython 3.12 - 3.14. python-version: "3.12" - name: Install Atheris run: pip install --require-hashes -r fuzz/requirements-fuzz.txt diff --git a/CHANGELOG.md b/CHANGELOG.md index 9313538b..512cc3ab 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -10,5 +10,6 @@ - 순수 영숫자 토큰은 정규식 호출을 건너뛰되 다국어·문장부호 토큰화 결과는 기존 의미와 동일하게 유지합니다. 근거, 한계, APA 7 참고문헌은 [`docs/doctoring/token-fast-path-equivalence.md`](docs/doctoring/token-fast-path-equivalence.md)에 기록했습니다. ### Fixed +- Python 3.14 coverage 검증이 설치할 수 없던 Atheris 3.0.0 lock을 3.1.0의 공식 Python 3.12–3.14 wheel digest로 갱신했습니다. - 단일·일괄 대상 크기 입력을 비웠을 때 이전 custom validity와 `aria-invalid` 상태를 즉시 초기화해 현재 필수 입력 상태를 정확히 전달합니다. - 업로드 파일명의 경로 구분자를 정규화하여 POSIX에서도 Windows 형식의 클라이언트 경로가 일관된 basename으로 기록되도록 수정했습니다. diff --git a/CLAUDE.md b/CLAUDE.md index cc7870fb..89e5d3a7 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -35,7 +35,7 @@ docker build -t codec-carver . && docker run -p 8000:8000 codec-carver # MCP server python mcp_driver.py -# Fuzzing (Atheris; CPython <= 3.12, not Windows) +# Fuzzing (Atheris; CPython 3.12-3.14, not Windows; CI uses 3.12) pip install --require-hashes -r fuzz/requirements-fuzz.txt python fuzz/fuzz_parse_silencedetect.py -max_total_time=60 fuzz/corpus/parse_silencedetect ``` diff --git a/fuzz/README.md b/fuzz/README.md index d12a61a4..c5287b50 100644 --- a/fuzz/README.md +++ b/fuzz/README.md @@ -12,7 +12,8 @@ in-process data structures. | [Atheris](https://github.com/google/atheris) | coverage-guided (libFuzzer) fuzzing | Apache-2.0 | | [Hypothesis](https://hypothesis.readthedocs.io/) | property-based tests in the normal suite | MPL-2.0 | -Both are permissive (no GPL/AGPL). Atheris supports CPython 3.6–3.12. +Both are permissive (no GPL/AGPL). The pinned Atheris artifacts support +CPython 3.12–3.14; CI uses Python 3.12 for deterministic fuzz runs. ## Targets diff --git a/fuzz/fuzz_build_segments.py b/fuzz/fuzz_build_segments.py index 1b4c1f86..2fdf12c8 100644 --- a/fuzz/fuzz_build_segments.py +++ b/fuzz/fuzz_build_segments.py @@ -15,7 +15,7 @@ Malformed inputs are expected to raise ``ValueError`` (a documented guard); any other exception is a defect. -Run locally (Python 3.8 - 3.12):: +Run locally (Python 3.12 - 3.14):: python fuzz/fuzz_build_segments.py -atheris_runs=200000 fuzz/corpus/build_segments """ diff --git a/fuzz/fuzz_parse_probe_payload.py b/fuzz/fuzz_parse_probe_payload.py index 34e79bc5..7f0b55a8 100644 --- a/fuzz/fuzz_parse_probe_payload.py +++ b/fuzz/fuzz_parse_probe_payload.py @@ -10,7 +10,7 @@ project's own ``MediaShrinkerError`` — never an unhandled ``KeyError`` / ``TypeError`` / ``ValueError``. -Run locally (Python 3.8 - 3.12):: +Run locally (Python 3.12 - 3.14):: python fuzz/fuzz_parse_probe_payload.py -atheris_runs=200000 fuzz/corpus/parse_probe_payload """ diff --git a/fuzz/fuzz_parse_silencedetect.py b/fuzz/fuzz_parse_silencedetect.py index a6c72bd4..db96820e 100644 --- a/fuzz/fuzz_parse_silencedetect.py +++ b/fuzz/fuzz_parse_silencedetect.py @@ -7,7 +7,7 @@ arbitrary byte strings and asserts the parser never raises and only ever produces well-formed, ordered silence intervals. -Run locally (Python 3.8 - 3.12):: +Run locally (Python 3.12 - 3.14):: python fuzz/fuzz_parse_silencedetect.py -atheris_runs=200000 fuzz/corpus/parse_silencedetect diff --git a/fuzz/requirements-fuzz.txt b/fuzz/requirements-fuzz.txt index 98f8cacf..9629c2d7 100644 --- a/fuzz/requirements-fuzz.txt +++ b/fuzz/requirements-fuzz.txt @@ -1,7 +1,6 @@ # Hash-pinned dependency for the coverage-guided fuzzing job (Atheris). # Regenerate with: uv pip compile fuzz/requirements-fuzz.in --generate-hashes -atheris==3.0.0 \ - --hash=sha256:1f0929c7bc3040f3fe4102e557718734190cf2d7718bbb8e3ce6d3eb56ef5bb3 \ - --hash=sha256:510e502c57b6dc615fb174066407af620d4c7f73cf08a782c86e7761bf12c4eb \ - --hash=sha256:8a5c8a781467c187da40fd29139784193e2647058831f837f675d0bb8cbd8746 \ - --hash=sha256:a402cdca8a650d1371050b1f9552eb4cdc488d2db64950d603c4560318365eac +atheris==3.1.0 \ + --hash=sha256:ec5e11f21a4c197fe91f7aea2b2de88e623c73a21fc07b105ac6329a1588457b \ + --hash=sha256:f8a9f51ce8369026e8eb7b7174835e8c4c85a1a6db5d9add36c15100779d2a39 \ + --hash=sha256:315a0b5c819852b1ffe1ca72efc389c7724881f2c33e4aacb8c6bcec49bd5011 diff --git a/tests/test_ci_workflow.py b/tests/test_ci_workflow.py index 2c3f2b13..848ff886 100644 --- a/tests/test_ci_workflow.py +++ b/tests/test_ci_workflow.py @@ -6,10 +6,29 @@ ROOT = Path(__file__).resolve().parents[1] CI_WORKFLOW = ROOT / ".github" / "workflows" / "ci.yml" +FUZZ_WORKFLOW = ROOT / ".github" / "workflows" / "fuzz.yml" +FUZZ_REQUIREMENTS = ROOT / "fuzz" / "requirements-fuzz.txt" class CiWorkflowTests(unittest.TestCase): - """Keep Rust CI reproducible on runners without a suitable default toolchain.""" + """Keep repository CI reproducible across its supported runner toolchains.""" + + def test_atheris_lock_supports_fuzz_and_coverage_python_versions(self) -> None: + """Pin artifacts installable by the Python 3.12 fuzz and 3.14 coverage lanes.""" + + requirements = FUZZ_REQUIREMENTS.read_text(encoding="utf-8") + workflow = FUZZ_WORKFLOW.read_text(encoding="utf-8") + + self.assertIn("atheris==3.1.0", requirements) + self.assertIn( + "sha256:ec5e11f21a4c197fe91f7aea2b2de88e623c73a21fc07b105ac6329a1588457b", + requirements, + ) + self.assertIn( + "sha256:315a0b5c819852b1ffe1ca72efc389c7724881f2c33e4aacb8c6bcec49bd5011", + requirements, + ) + self.assertIn("CPython 3.12 - 3.14", workflow) def test_rust_job_installs_and_uses_rust_1_88_with_rustfmt(self) -> None: """Require edition-2024 Rust and rustfmt before formatting or tests run."""