diff --git a/.github/workflows/release-podcast-tests.yml b/.github/workflows/release-podcast-tests.yml new file mode 100644 index 000000000..6d93ebb63 --- /dev/null +++ b/.github/workflows/release-podcast-tests.yml @@ -0,0 +1,29 @@ +name: Release podcast offline tests +on: + pull_request: + paths: + - release/podcast/** + - .github/workflows/release-podcast*.yml + push: + branches: [master] + paths: + - release/podcast/** + - .github/workflows/release-podcast*.yml +permissions: + contents: read +jobs: + test: + runs-on: ubuntu-latest + timeout-minutes: 5 + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7 + with: + persist-credentials: false + - uses: actions/setup-go@924ae3a1cded613372ab5595356fb5720e22ba16 # v6 + with: + go-version-file: go.mod + - name: Install tools for the offline MP3 decode test + run: | + sudo apt-get update + sudo apt-get install -y --no-install-recommends ffmpeg + - run: go test -race -count=1 -v ./release/podcast diff --git a/.github/workflows/release-podcast.yml b/.github/workflows/release-podcast.yml new file mode 100644 index 000000000..d25597e34 --- /dev/null +++ b/.github/workflows/release-podcast.yml @@ -0,0 +1,126 @@ +name: Release podcast + +on: + release: + types: [published] + workflow_dispatch: + inputs: + from: + description: Published version, or inclusive first version of a range + default: v0.64.0 + required: true + type: string + to: + description: Optional inclusive last version (at most three releases) + default: v0.64.2 + required: false + type: string + mode: + description: Validate is free; script and audio use OpenAI + default: validate + type: choice + options: [validate, script, audio] + include_prereleases: + description: Allow published prereleases (manual only) + default: false + type: boolean + force_regenerate: + description: Allow another paid attempt even if reserved before + default: false + type: boolean + +permissions: {} + +# Serialize the artifact lookup and reservation, including overlapping rollups. +# GitHub may replace an older pending run; recover that run with manual dispatch. +concurrency: + group: release-podcast-${{ github.repository }} + cancel-in-progress: false + +jobs: + generate: + if: >- + github.repository == 'navidrome/navidrome' && + ((github.event_name == 'release' && !github.event.release.draft && !github.event.release.prerelease && vars.RELEASE_AUDIO_ENABLED == 'true') || + (github.event_name == 'workflow_dispatch' && github.ref == format('refs/heads/{0}', github.event.repository.default_branch))) + runs-on: ubuntu-latest + timeout-minutes: 10 + permissions: + contents: read + actions: read + env: + AUDIO_MODE: ${{ github.event_name == 'release' && 'audio' || inputs.mode }} + AUDIO_ENABLED: ${{ vars.RELEASE_AUDIO_ENABLED }} + AUDIO_TEXT_MODEL: ${{ vars.RELEASE_AUDIO_TEXT_MODEL }} + AUDIO_TTS_MODEL: ${{ vars.RELEASE_AUDIO_TTS_MODEL }} + AUDIO_VOICE: ${{ vars.RELEASE_AUDIO_VOICE }} + steps: + # Always execute the reviewed default-branch helper, never release-note data. + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7 + with: + ref: ${{ github.event.repository.default_branch }} + persist-credentials: false + - uses: actions/setup-go@924ae3a1cded613372ab5595356fb5720e22ba16 # v6 + with: + go-version-file: go.mod + - name: Build the shared Go CLI before exposing credentials + run: go build -o "$RUNNER_TEMP/release-podcast" ./release/podcast + - name: Install MP3 inspection tools before generation + if: env.AUDIO_MODE == 'audio' && env.AUDIO_ENABLED == 'true' + run: | + sudo apt-get update + sudo apt-get install -y --no-install-recommends ffmpeg + - name: Resolve sources and check limits and duplicate attempts + id: prepare + env: + GH_TOKEN: ${{ github.token }} + AUDIO_KEY_CONFIGURED: ${{ secrets.OPENAI_API_KEY != '' }} + run: '"$RUNNER_TEMP/release-podcast" --github-stage prepare' + - name: Reserve this paid attempt before contacting OpenAI + if: steps.prepare.outputs.generate == 'true' + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 + with: + # Exact deterministic name supports API filtering across runs. + # Keep immutable within a run; force requires a fresh dispatch. + name: ${{ steps.prepare.outputs.reservation }} + path: release-podcast/manifest.json + if-no-files-found: error + retention-days: 90 + - name: Generate and validate the grounded script + if: steps.prepare.outputs.generate == 'true' + env: + OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }} + GH_TOKEN: ${{ github.token }} + run: '"$RUNNER_TEMP/release-podcast" --github-stage script' + - name: Checkpoint the validated script + if: steps.prepare.outputs.generate == 'true' + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 + with: + name: release-podcast-script-${{ github.run_id }}-${{ github.run_attempt }} + path: | + release-podcast/transcript.txt + release-podcast/sources.json + release-podcast/evidence.json + release-podcast/manifest.json + if-no-files-found: error + retention-days: 30 + - name: Synthesize and inspect the MP3 + if: steps.prepare.outputs.generate == 'true' && env.AUDIO_MODE == 'audio' + env: + OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }} + GH_TOKEN: ${{ github.token }} + run: '"$RUNNER_TEMP/release-podcast" --github-stage speech' + - name: Upload review artifacts + if: always() && steps.prepare.outputs.prepared == 'true' + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 + with: + name: release-podcast-${{ github.run_id }}-${{ github.run_attempt }} + path: | + release-podcast/transcript.txt + release-podcast/release-podcast.mp3 + release-podcast/sources.json + release-podcast/evidence.json + release-podcast/manifest.json + if-no-files-found: error + retention-days: 30 + compression-level: 0 diff --git a/release/podcast/README.md b/release/podcast/README.md new file mode 100644 index 000000000..f651977a6 --- /dev/null +++ b/release/podcast/README.md @@ -0,0 +1,246 @@ +# Release podcast + +A standalone Go CLI turns **published Navidrome release notes** into a grounded +English recap and an MP3. The local CLI and GitHub workflow share the same source +resolution, evidence checks, cost limits and generation stages. Outputs are +`transcript.txt`, `release-podcast.mp3`, `sources.json`, `evidence.json` and +`manifest.json`. Nothing is uploaded to release assets or social media. + +## Local use, including before merge + +From a checkout of this PR, with the Go version from `go.mod`: + +```sh +go run ./release/podcast --from 0.64.0 --to 0.64.2 \ + --dry-run --output /tmp/navidrome-podcast-validation +``` + +This resolves the inclusive range of published releases, writes exact sources +and a manifest, and makes **zero OpenAI requests**. It needs GitHub network +access; `GH_TOKEN` is optional for public notes and increases the rate limit. +It needs neither an OpenAI key nor repository variables, and does not spoof +GitHub context. To select one published release, use `--from` alone: + +```sh +go run ./release/podcast --from v0.64.2 \ + --mode validate --output /tmp/navidrome-podcast-validation +``` + +Select one release or an inclusive range of at most three releases. Versions +may omit the `v` prefix. There is no comma-separated tag-list option. +Ranges require ordered stable-version endpoints that both exist as published +releases. Listing is bounded to 1,000 release records; larger listings require +single-release selection with `--from` alone. Published prereleases require +`--include-prereleases`; select a prerelease endpoint with `--from` alone. +Drafts are discarded. +Prereleases use SemVer precedence, including numeric identifiers: `rc.2` +precedes `rc.10`, and both precede the corresponding stable release. Thus a +stable lower range bound excludes its own RCs; a stable upper bound includes +its RCs only with `--include-prereleases`. + +When you separately decide to incur API usage, make `OPENAI_API_KEY` available +in your local process environment through your own secure setup. **Never put +the key in a CLI argument, log, commit, or transcript.** A repository Actions +secret is not a local environment variable; this tool does not retrieve it. +Choose supported values for the nonsecret `TEXT_MODEL`, `TTS_MODEL` and `VOICE` +variables used in this example: + +```sh +go run ./release/podcast --from 0.64.0 --to 0.64.2 \ + --mode audio --allow-paid \ + --text-model "$TEXT_MODEL" --tts-model "$TTS_MODEL" --voice "$VOICE" \ + --output /tmp/navidrome-podcast-preview +``` + +Install `ffmpeg` (including `ffprobe`) yourself before local audio mode. Media +preflight runs before any paid request. Then listen to +`/tmp/navidrome-podcast-preview/release-podcast.mp3` and compare the transcript +and evidence with the notes. Local preview does **not** require merging the PR +or setting `RELEASE_AUDIO_ENABLED`. + +`--mode script --allow-paid --text-model MODEL` generates only the transcript +and evidence. `--dry-run` always forces validation, even if `--mode audio` or +`--allow-paid` is also present. CLI model/voice flags override `AUDIO_TEXT_MODEL`, +`AUDIO_TTS_MODEL` and `AUDIO_VOICE` environment values; there are no model/voice +defaults. No API host, repository, key, shell command or pricing override is +accepted as a CLI option. + +Use a fresh output directory for each intentional preview. Paid runs acquire +an exclusive local `.release-podcast.lock` and persist a reservation in +`.attempts/` **before** contacting OpenAI. Failed attempts also remain reserved. +`--force` permits another paid attempt and replacement of generated files; +it cannot bypass an active lock. After a killed process, review the output and +ledger before manually removing a stale lock. Validation refuses directories +containing generated script/audio files so it cannot relabel older audio. +Local deduplication is scoped to the output directory; a new directory or deleted +ledger can permit another charge. Script and audio modes have distinct keys. + +## Models and voice + +These are supported choices, **not finalized user defaults**: + +| Setting | Reviewed options | +| --- | --- | +| Text model | `gpt-6-luna` (low reasoning), `gpt-4.1-mini-2025-04-14` | +| Speech model | `gpt-4o-mini-tts-2025-12-15`, `tts-1`, `tts-1-hd` | +| Legacy voices | `alloy`, `echo`, `fable`, `onyx`, `nova`, `shimmer` | +| Mini TTS voices | Legacy voices plus `ash`, `ballad`, `coral`, `sage`, `verse`, `marin`, `cedar` | + +Onyx and cedar are candidates to audition for a male-style delivery; perceived +voice gender is subjective. Mini TTS receives fixed calm English instructions; +legacy TTS does not. Unsupported models fail before paid requests. A new model +requires a code/pricing review. Luna is an API alias whose behavior can change; +the other script model and Mini TTS use pinned snapshots. + +## GitHub Actions setup and publication + +The workflow is `.github/workflows/release-podcast.yml`, named **Release +podcast**. Automatic generation is disabled by default. The secret and variable +names already communicated during planning are retained, so no configuration +rename is required: + +1. The maintainer adds repository Actions secret `OPENAI_API_KEY` personally. + One project key covers Responses and speech when its model/endpoint access + allows both. Use a dedicated project key and usage alerts. +2. Choose repository variables `RELEASE_AUDIO_TEXT_MODEL`, + `RELEASE_AUDIO_TTS_MODEL` and `RELEASE_AUDIO_VOICE`. +3. Set `RELEASE_AUDIO_ENABLED` to literal `true` only when paid workflow runs + are authorized. Blank variables are fine for offline tests/manual validation. +4. After merge, dispatch **Release podcast** from `master`, starting with + `validate`, `from=v0.64.0` and `to=v0.64.2`. Clear `to` to select only `from`. + Models and voice come from repository + variables, not manual inputs. No named GitHub Environment is configured. + +The automatic trigger is **`release: published`**, stable releases only. It +fetches the event's exact release ID, never `/releases/latest`. Tags, edits, +drafts and prereleases do not automatically generate audio. Promotion of an +already-published prerelease may require manual dispatch; manual prereleases +need explicit opt-in. + +The existing `pipeline.yml` uses `GITHUB_TOKEN` for GoReleaser and +`release/goreleaser.yml` sets `draft: true`. Those behaviors are unchanged. +Maintainer publication of the draft through GitHub can trigger this workflow; +publication with `GITHUB_TOKEN` generally suppresses downstream release events. +Use manual dispatch after such publication, or separately review an explicit +dispatch integration with narrow Actions permissions. Do not add a broad PAT +or change draft publishing to solve this integration. + +Land the workflow before the next tag. Release events are associated with the +tagged commit; historical tags cannot be assumed to contain a new workflow. +Manual dispatch requires the workflow on the default branch and deliberately +rejects other branches. Checkout executes the reviewed default-branch helper, +with credentials not persisted. **Use the first-class local CLI to test the PR +before merge**, rather than bypassing this Actions guard. + +The workflow builds the shared Go binary and installs media tools in +credential-free steps. It exposes `OPENAI_API_KEY` only to generation steps, +grants `contents: read` and `actions: read`, and pins actions to verified SHAs. +No PR event can run the paid workflow. The offline PR test workflow has no key. + +| Mode | Maximum OpenAI requests | Outputs | +| --- | --- | --- | +| `validate` / `--dry-run` | None | Sources and manifest; modeled text estimate if configured | +| `script` | One Responses request | Transcript, evidence, sources and manifest | +| `audio` | One Responses plus one speech request | Script outputs and validated MP3 | + +## Limits, grounding and costs + +Target about two minutes: 250–280 total words and 105–145 seconds. Small releases +may be shorter; correctness takes priority over filler. There are no application +or SDK retries/repair calls. Timeouts may already be billed. HTTP requests have +a one-minute timeout, media inspection 30 seconds, and the CLI/workflow ten +minutes. Sources and the complete prompt each have separate 64 KiB caps; +oversized material fails without truncating warnings. Text output is capped at +3,000 tokens including reasoning/evidence. Narration is at most 280 words and +2,500 characters. Mini TTS additionally uses a conservative 2,000 UTF-8-byte +input ceiling including instructions, to stay below its input-token limit. + +The model has no tools and receives only public notes as untrusted evidence, +not executable instructions. No embedded links or draft advisories are fetched. +Strict JSON maps sentences to exact source excerpts and required cautions. +IDs/excerpts and migration/security coverage must match; lexical checks retain +the prototype's backup, client resync, experimental/opt-in, plugin networking, +Docker discovery, and slow-storage/32-bit scan cautions. These checks do not +prove semantic entailment or classify language perfectly. **Human factual and +listening review remains required.** Every transcript discloses the AI voice. +`store: false` does not imply zero provider retention. + +Reviewed prices, USD per million units, checked 2026-10-01: + +| Model | Input | Output | +| --- | ---: | ---: | +| GPT-6 Luna | $0.10/text token | $0.50/text token | +| GPT-4.1 mini | $0.40/text token | $1.60/text token | +| TTS-1 | $15/character | — | +| TTS-1 HD | $30/character | — | +| GPT-4o mini TTS | $0.60/text token | $12/audio token | + +Preflight rejects a **modeled allowance above $0.10**, using all serialized +prompt bytes plus 1,024 framing tokens and maximum text output. Legacy TTS uses +the character cap. Mini TTS uses a conservative **6,000 audio-token allowance** +plus input; its API offers no enforceable output-token/dollar cap, so this is +an estimate, **not a billing guarantee**. Prices, anomalous audio duration, +taxes, GitHub usage and deliberate later runs can change costs. Project budget +alerts are soft thresholds. The original approximately $0.03 example applied +to illustrative 4.1 mini + TTS-1 inputs, not every supported model combination. +Neither this PR nor an estimate authorizes a paid prototype run. + +## Actions duplicate attempts and recovery + +Jobs serialize separately from the build pipeline. Before paid requests, the +GitHub API's exact `name` filter looks up the deterministic reservation name; +unrelated repository artifacts do not enter pagination. A bounded lookup of up +to 10,000 matching reservations fails closed on errors or overflow. A reservation +artifact is uploaded first, keyed by repository, sorted release IDs and mode. +Matching attempts are skipped even after failure or +source/model edits; explicit manual `force_regenerate=true` permits another +paid attempt. Rollups and single releases are different source sets. +Use a fresh manual dispatch for forced generation. Reservation names repeat +across runs but are immutable within a run: a rerun of an already-reserved +forced attempt fails at upload before any paid request, preserving the ledger. + +Reservations last 90 days, review outputs/script checkpoints 30 days, limited +by repository policy. Deleted/expired artifacts permit another attempt, so +this is best-effort deduplication, not a permanent billing ledger. GitHub can +replace an older pending concurrency run; dispatch it manually if needed. + +Both paid stages re-fetch notes and stop if changed/withdrawn. Scripts are +checkpointed before TTS. Invalid scripts never reach speech, and error JSON +never becomes MP3. `ffprobe` checks codec/duration and `ffmpeg` decodes the full +file, restricted to MP3 and local file/pipe protocols. Duration deviations are +flagged without regeneration. Manifests record source/prompt/script/audio hashes, +the generation fingerprint, commits, config, request counts, text usage and +duration. Child media/git commands do not inherit API credentials. + +Do not force regeneration to fix delivery. If the runner/checkpoint is lost, +automatic cross-run checkpoint recovery is not implemented: review the saved +script and decide whether a new deliberate attempt is warranted. It may be +billed. Actions artifacts expire and require signed-in repository read access; +they are not permanent anonymous podcast URLs. + +## Offline verification + +```sh +go test -race -count=1 -v ./release/podcast +``` + +HTTP transports are mocked and unexpected requests fail. Public v0.64.0/.1/.2 +fixtures test omissions, unsafe versions/output, inclusive ranges, paid guards, +local locking/reservations, changed sources/checkpoints, Actions trust/config, +artifact expiry/lookup limits, budget caps, redirects and no retries. Real MP3 +decoding uses a local synthetic tone with mocked OpenAI. CI installs media +tools so that test runs rather than skips. The tool uses Go's standard library; +no Python implementation or additional Go module dependency is required. + +## References + +- [GitHub release events](https://docs.github.com/en/actions/reference/workflows-and-actions/events-that-trigger-workflows#release) +- [GITHUB_TOKEN event suppression](https://docs.github.com/en/actions/how-tos/writing-workflows/choosing-when-your-workflow-runs/triggering-a-workflow) +- [GitHub exact-name artifact filter](https://docs.github.com/en/rest/actions/artifacts#list-artifacts-for-a-repository) +- [SemVer prerelease precedence](https://semver.org/) +- [GPT-6 Luna](https://developers.openai.com/api/docs/models/gpt-6-luna) +- [GPT-4.1 mini](https://developers.openai.com/api/docs/models/gpt-4.1-mini) +- [Mini TTS](https://developers.openai.com/api/docs/models/gpt-4o-mini-tts) +- [TTS-1](https://developers.openai.com/api/docs/models/tts-1) +- [TTS-1 HD](https://developers.openai.com/api/docs/models/tts-1-hd) +- [Speech guide](https://developers.openai.com/api/docs/guides/text-to-speech) diff --git a/release/podcast/actions.go b/release/podcast/actions.go new file mode 100644 index 000000000..66f890663 --- /dev/null +++ b/release/podcast/actions.go @@ -0,0 +1,177 @@ +package main + +import ( + "context" + "encoding/json" + "errors" + "fmt" + "net/url" + "os" +) + +type event struct { + Action string `json:"action"` + Release releaseRecord `json:"release"` + Repository struct { + DefaultBranch string `json:"default_branch"` + } `json:"repository"` + Inputs struct { + From string `json:"from"` + To string `json:"to"` + Mode string `json:"mode"` + IncludePrereleases string `json:"include_prereleases"` + Force string `json:"force_regenerate"` + } `json:"inputs"` +} + +func githubEvent() (event, error) { + var ev event + if os.Getenv("GITHUB_ACTIONS") != "true" || os.Getenv("GITHUB_REPOSITORY") != repository { + return ev, errors.New("Actions entrypoint requires the navidrome repository context") + } + data, err := os.ReadFile(os.Getenv("GITHUB_EVENT_PATH")) // #nosec G703 -- event path is provided by the verified GitHub runner context, never release data. + if err != nil || len(data) > 8<<20 || json.Unmarshal(data, &ev) != nil { + return ev, errors.New("invalid Actions event") + } + switch os.Getenv("GITHUB_EVENT_NAME") { + case "workflow_dispatch": + if ev.Repository.DefaultBranch == "" || os.Getenv("GITHUB_REF") != "refs/heads/"+ev.Repository.DefaultBranch { + return ev, errors.New("manual Actions generation must use the default branch") + } + case "release": + if ev.Action != "published" { + return ev, errors.New("expected a release published event") + } + if _, err := normalizeRelease(ev.Release, false); err != nil { + return ev, err + } + default: + return ev, errors.New("unsupported Actions event") + } + return ev, nil +} + +func (e *engine) duplicateArtifact(ctx context.Context, reservation string) (bool, error) { + for page := 1; page <= 100; page++ { + var response struct { + Artifacts []struct { + Name string `json:"name"` + Expired bool `json:"expired"` + } `json:"artifacts"` + } + if err := e.github(ctx, fmt.Sprintf("/actions/artifacts?name=%s&per_page=100&page=%d", url.QueryEscape(reservation), page), &response); err != nil { + return false, err + } + for _, a := range response.Artifacts { + if !a.Expired && a.Name == reservation { + return true, nil + } + } + if len(response.Artifacts) < 100 { + return false, nil + } + } + return false, errors.New("matching reservation ledger exceeds lookup limit; manual review required") +} + +func actionsOutput(key, value string) error { + f, err := os.OpenFile(os.Getenv("GITHUB_OUTPUT"), os.O_APPEND|os.O_WRONLY, 0600) // #nosec G703 -- append only to the runner-provided Actions output file. + if err != nil { + return errors.New("Actions output file unavailable") + } + defer f.Close() + if _, err := fmt.Fprintf(f, "%s=%s\n", key, value); err != nil { + return errors.New("cannot write Actions output") + } + return nil +} + +func (e *engine) runGitHub(ctx context.Context, stage string) error { + ev, err := githubEvent() + if err != nil { + return err + } + switch stage { + case "script": + return e.generateScript(ctx) + case "speech": + return e.generateSpeech(ctx) + case "prepare": + return e.prepareGitHub(ctx, ev) + default: + return errors.New("invalid Actions stage") + } +} + +func (e *engine) prepareGitHub(ctx context.Context, ev event) error { + manual := os.Getenv("GITHUB_EVENT_NAME") == "workflow_dispatch" + mode := "audio" + allow, force := false, false + if manual { + mode = ev.Inputs.Mode + if mode == "" { + mode = "validate" + } + allow = ev.Inputs.IncludePrereleases == "true" + force = ev.Inputs.Force == "true" + } + if mode != "validate" && mode != "script" && mode != "audio" { + return errors.New("invalid manual mode") + } + if mode != "validate" && (os.Getenv("AUDIO_ENABLED") != "true" || os.Getenv("AUDIO_KEY_CONFIGURED") != "true") { + return errors.New("paid Actions modes require explicit enablement and OPENAI_API_KEY in Secrets") + } + c, err := reviewedConfig(os.Getenv("AUDIO_TEXT_MODEL"), os.Getenv("AUDIO_TTS_MODEL"), os.Getenv("AUDIO_VOICE"), mode) + if err != nil { + return err + } + if mode == "audio" { + if err := e.mediaTools(ctx); err != nil { + return err + } + } + var sources []source + if manual { + sources, err = e.resolveLocal(ctx, options{from: ev.Inputs.From, to: ev.Inputs.To, includePrereleases: allow}) + } else { + var record releaseRecord + err = e.github(ctx, fmt.Sprintf("/releases/%d", ev.Release.ID), &record) + if err == nil { + var s source + s, err = normalizeRelease(record, false) + if err == nil && (s.ID != ev.Release.ID || s.Tag != ev.Release.Tag) { + err = errors.New("release identity changed") + } + sources = []source{s} + } + } + if err != nil { + return err + } + m, err := e.prepare(ctx, sources, c, mode, "github", allow, force) + if err != nil { + return err + } + m.RunID = os.Getenv("GITHUB_RUN_ID") + m.RunAttempt = os.Getenv("GITHUB_RUN_ATTEMPT") + m.EventSHA = os.Getenv("GITHUB_SHA") + if mode != "validate" { + repeated, err := e.duplicateArtifact(ctx, m.Reservation) + if err != nil { + return err + } + if repeated && !force { + m.Status = "duplicate" + } + } + if err := e.saveSources(sources, m); err != nil { + return err + } + if err := actionsOutput("reservation", m.Reservation); err != nil { + return err + } + if err := actionsOutput("prepared", "true"); err != nil { + return err + } + return actionsOutput("generate", fmt.Sprint(mode != "validate" && m.Status != "duplicate")) +} diff --git a/release/podcast/main.go b/release/podcast/main.go new file mode 100644 index 000000000..942e88d73 --- /dev/null +++ b/release/podcast/main.go @@ -0,0 +1,181 @@ +// release-podcast is a standalone, artifact-only release narration tool. +package main + +import ( + "context" + "embed" + "errors" + "flag" + "fmt" + "io" + "net/http" + "os" + "os/exec" + "path/filepath" + "strings" + "time" +) + +//go:embed prompt.txt +var promptFiles embed.FS + +const repository = "navidrome/navidrome" + +type options struct { + from, to, mode, textModel, ttsModel, voice, output, githubStage string + allowPaid, dryRun, includePrereleases, force bool +} + +func parseOptions(args []string, stderr io.Writer) (options, error) { + var o options + f := flag.NewFlagSet("release-podcast", flag.ContinueOnError) + f.SetOutput(stderr) + f.StringVar(&o.from, "from", "", "Published version to select, or inclusive first version of a range (v prefix optional)") + f.StringVar(&o.to, "to", "", "Optional inclusive last published version of a range") + f.StringVar(&o.mode, "mode", "validate", "validate (free), script, or audio") + f.StringVar(&o.textModel, "text-model", os.Getenv("AUDIO_TEXT_MODEL"), "Reviewed script model; no default") + f.StringVar(&o.ttsModel, "tts-model", os.Getenv("AUDIO_TTS_MODEL"), "Reviewed speech model; no default") + f.StringVar(&o.voice, "voice", os.Getenv("AUDIO_VOICE"), "Compatible stock voice; no default") + f.StringVar(&o.output, "output", "release-podcast", "Output directory (sources, manifest, script and MP3)") + f.BoolVar(&o.dryRun, "dry-run", false, "Force validate mode; never contact OpenAI") + f.BoolVar(&o.allowPaid, "allow-paid", false, "Explicitly permit bounded OpenAI requests for this local invocation") + f.BoolVar(&o.includePrereleases, "include-prereleases", false, "Include published prereleases") + f.BoolVar(&o.force, "force", false, "Permit another paid attempt and replacement of generated files") + f.StringVar(&o.githubStage, "github-stage", "", "Actions entrypoint: prepare, script, or speech") + if err := f.Parse(args); err != nil { + return o, err + } + if f.NArg() != 0 { + return o, errors.New("unexpected positional arguments") + } + if o.dryRun { + o.mode = "validate" + } + if o.mode != "validate" && o.mode != "script" && o.mode != "audio" { + return o, errors.New("mode must be validate, script, or audio") + } + if o.githubStage != "" && (o.from != "" || o.to != "" || o.allowPaid || o.dryRun || o.force || o.includePrereleases) { + return o, errors.New("Actions stages use trusted event/environment configuration, not local selection flags") + } + return o, nil +} + +type engine struct { + client *http.Client + githubToken, openAIKey, output, prompt string + command func(context.Context, string, ...string) ([]byte, error) +} + +func newEngine(output string) (*engine, error) { + abs, err := filepath.Abs(output) + if err != nil { + return nil, errors.New("invalid output directory") + } + prompt, err := promptFiles.ReadFile("prompt.txt") + if err != nil { + return nil, errors.New("embedded prompt unavailable") + } + return &engine{output: abs, prompt: string(prompt), githubToken: os.Getenv("GH_TOKEN"), openAIKey: os.Getenv("OPENAI_API_KEY"), + client: &http.Client{Timeout: time.Minute, CheckRedirect: func(*http.Request, []*http.Request) error { return http.ErrUseLastResponse }}, + command: func(ctx context.Context, name string, args ...string) ([]byte, error) { + if name != "git" && name != "ffprobe" && name != "ffmpeg" { + return nil, errors.New("unsupported local tool") + } + cmd := exec.CommandContext(ctx, name, args...) // #nosec G204 -- fixed tool allowlist; arguments are data, never shell source. + for _, item := range os.Environ() { + if !strings.HasPrefix(item, "OPENAI_API_KEY=") && !strings.HasPrefix(item, "GH_TOKEN=") { + cmd.Env = append(cmd.Env, item) + } + } + return cmd.Output() + }, + }, nil +} + +func run(ctx context.Context, args []string, stdout, stderr io.Writer) error { + o, err := parseOptions(args, stderr) + if err != nil { + return err + } + e, err := newEngine(o.output) + if err != nil { + return err + } + if o.githubStage != "" { + return e.runGitHub(ctx, o.githubStage) + } + if os.Getenv("GITHUB_ACTIONS") == "true" { + return errors.New("Actions must use the guarded github-stage entrypoint") + } + return e.runLocal(ctx, o, stdout) +} + +func (e *engine) runLocal(ctx context.Context, o options, stdout io.Writer) error { + if o.mode != "validate" && (!o.allowPaid || e.openAIKey == "") { + return errors.New("paid local modes require --allow-paid and OPENAI_API_KEY in the environment") + } + cfg, err := reviewedConfig(o.textModel, o.ttsModel, o.voice, o.mode) + if err != nil { + return err + } + if o.mode == "audio" { + if err := e.mediaTools(ctx); err != nil { + return err + } + } + sources, err := e.resolveLocal(ctx, o) + if err != nil { + return err + } + m, err := e.prepare(ctx, sources, cfg, o.mode, "local", o.includePrereleases, o.force) + if err != nil { + return err + } + if o.mode == "validate" { + if err := e.checkGeneratedFiles(); err != nil { + return err + } + if err := e.saveSources(sources, m); err != nil { + return err + } + _, err = fmt.Fprintln(stdout, "Validated published sources; OpenAI requests: 0. Outputs:", e.output) + return err + } + releaseLock, err := e.reserveLocal(m, o.force) + if err != nil { + return err + } + defer releaseLock() + if o.force { + for _, name := range []string{"transcript.txt", "release-podcast.mp3", "evidence.json"} { + if err := os.Remove(filepath.Join(e.output, name)); err != nil && !errors.Is(err, os.ErrNotExist) { + return errors.New("cannot replace previous generated outputs") + } + } + } + if err := e.saveSources(sources, m); err != nil { + return err + } + if err := e.generateScript(ctx); err != nil { + return err + } + if o.mode == "audio" { + if err := e.generateSpeech(ctx); err != nil { + return err + } + } + _, err = fmt.Fprintln(stdout, "Generated review outputs:", e.output) + return err +} + +func main() { + ctx, cancel := context.WithTimeout(context.Background(), 10*time.Minute) + defer cancel() + if err := run(ctx, os.Args[1:], os.Stdout, os.Stderr); err != nil { + if errors.Is(err, flag.ErrHelp) { + return + } + fmt.Fprintln(os.Stderr, "Release podcast:", err) + os.Exit(1) + } +} diff --git a/release/podcast/pipeline.go b/release/podcast/pipeline.go new file mode 100644 index 000000000..f77b5f9a4 --- /dev/null +++ b/release/podcast/pipeline.go @@ -0,0 +1,471 @@ +package main + +import ( + "bytes" + "context" + "crypto/sha256" + "encoding/hex" + "encoding/json" + "errors" + "fmt" + "io" + "math" + "os" + "path/filepath" + "slices" + "sort" + "strconv" + "strings" + "time" +) + +const ( + maxPromptBytes = 65536 + maxOutputTokens = 3000 + maxScriptChars = 2500 + miniInputBytes = 2000 + modeledAudioTokens = 6000 + maxCostUSD = 0.10 + speechInstructions = "Speak in clear, calm English with a warm, measured delivery. Read API as letters." + intro = "Welcome to the Navidrome release recap. This narration uses an AI-generated voice." + closing = "For the full details and upgrade guidance, read the official release notes linked alongside this transcript." +) + +type modelConfig struct { + TextModel string `json:"text_model"` + TTSModel string `json:"tts_model"` + Voice string `json:"voice"` + Speed float64 `json:"speed"` + Instructions string `json:"speech_instructions"` +} + +var textRates = map[string][2]float64{"gpt-6-luna": {0.10, 0.50}, "gpt-4.1-mini-2025-04-14": {0.40, 1.60}} +var speechRates = map[string]float64{"tts-1": 15, "tts-1-hd": 30, "gpt-4o-mini-tts-2025-12-15": 0} + +func reviewedConfig(text, tts, voice, mode string) (modelConfig, error) { + c := modelConfig{TextModel: text, TTSModel: tts, Voice: voice, Speed: 1} + if _, ok := textRates[text]; !ok && (mode != "validate" || text != "") { + return c, errors.New("select a reviewed text model; no default is selected") + } + if _, ok := speechRates[tts]; !ok && (mode == "audio" || tts != "") { + return c, errors.New("select a reviewed speech model") + } + voices := ",alloy,echo,fable,onyx,nova,shimmer," + if tts == "gpt-4o-mini-tts-2025-12-15" { + voices += "ash,ballad,coral,sage,verse,marin,cedar," + c.Instructions = speechInstructions + } + if (voice != "" || mode == "audio") && (voice == "" || !slices.Contains(strings.Split(strings.Trim(voices, ","), ","), voice)) { + return c, errors.New("select a compatible reviewed stock voice") + } + return c, nil +} + +func narrationByteLimit(c modelConfig) int { + if c.Instructions != "" { + return miniInputBytes - len(c.Instructions) + } + return maxScriptChars * 4 // Character-priced models; Unicode still checked separately. +} + +type manifest struct { + Version int `json:"version"` + Mode string `json:"mode"` + Origin string `json:"origin"` + Config modelConfig `json:"config"` + Cost *float64 `json:"modeled_cost_usd"` + SourceHash string `json:"source_sha256"` + PromptHash string `json:"prompt_sha256"` + GenerationHash string `json:"generation_sha256"` + ImplementationSHA string `json:"implementation_sha"` + EventSHA string `json:"event_sha,omitempty"` + RunID string `json:"run_id,omitempty"` + RunAttempt string `json:"run_attempt,omitempty"` + CreatedAt string `json:"created_at"` + Reservation string `json:"reservation"` + TextRequests int `json:"text_requests"` + SpeechRequests int `json:"speech_requests"` + IncludePrereleases bool `json:"include_prereleases"` + Force bool `json:"force_regenerate"` + Status string `json:"status"` + ScriptHash string `json:"script_sha256,omitempty"` + AudioHash string `json:"audio_sha256,omitempty"` + Words int `json:"word_count,omitempty"` + Characters int `json:"character_count,omitempty"` + Duration float64 `json:"duration_seconds,omitempty"` + DurationNeedsReview bool `json:"duration_needs_review,omitempty"` + TextUsage json.RawMessage `json:"text_usage,omitempty"` +} + +func hashBytes(data []byte) string { h := sha256.Sum256(data); return hex.EncodeToString(h[:]) } +func hashJSON(value any) string { data, _ := json.Marshal(value); return hashBytes(data) } + +func (e *engine) textPayload(sources []source, c modelConfig) map[string]any { + input, _ := json.Marshal(map[string]any{"sources": promptSources(sources), "required_cautions": cautions(sources), + "introduction": intro, "closing": closing, "narration_limits": map[string]int{"max_words": 280, "max_characters": maxScriptChars, "max_utf8_bytes": narrationByteLimit(c)}}) + p := map[string]any{"model": c.TextModel, "store": false, "max_output_tokens": maxOutputTokens, "instructions": e.prompt, "input": string(input), + "text": map[string]any{"format": map[string]any{"type": "json_schema", "name": "release_narration", "strict": true, "schema": narrationSchema()}}} + if c.TextModel == "gpt-6-luna" { + p["reasoning"] = map[string]string{"effort": "low"} + } + return p +} + +func modeledCost(payload any, c modelConfig, mode string) (*float64, error) { + data, err := json.Marshal(payload) + if err != nil { + return nil, errors.New("cannot encode prompt") + } + if len(data) > maxPromptBytes { + return nil, errors.New("complete prompt exceeds byte limit; never truncate warnings") + } + rate, ok := textRates[c.TextModel] + if !ok { + return nil, nil + } + // Conservative byte-level input-token count, plus protocol framing. + cost := (float64(len(data)+1024)*rate[0] + maxOutputTokens*rate[1]) / 1e6 + if mode == "audio" { + if c.Instructions != "" { + cost += (float64(miniInputBytes)*0.60 + modeledAudioTokens*12) / 1e6 + } else { + cost += maxScriptChars * speechRates[c.TTSModel] / 1e6 + } + } + if cost > maxCostUSD { + return nil, errors.New("modeled cost exceeds $0.10; review source size or models") + } + return &cost, nil +} + +func (e *engine) prepare(ctx context.Context, sources []source, c modelConfig, mode, origin string, allow, force bool) (manifest, error) { + var m manifest + seen := map[int64]bool{} + total := 0 + ids := []int64{} + if len(sources) < 1 || len(sources) > 3 { + return m, errors.New("select one to three releases") + } + for _, s := range sources { + if seen[s.ID] { + return m, errors.New("duplicate release IDs") + } + seen[s.ID] = true + ids = append(ids, s.ID) + total += len(s.Body) + } + if total > maxSourceBytes { + return m, errors.New("combined notes exceed byte limit") + } + cost, err := modeledCost(e.textPayload(sources, c), c, mode) + if err != nil { + return m, err + } + sha, err := e.command(ctx, "git", "rev-parse", "HEAD") + if err != nil { + sha = []byte("unknown") + } + sort.Slice(ids, func(i, j int) bool { return ids[i] < ids[j] }) + m = manifest{Version: 2, Mode: mode, Origin: origin, Config: c, Cost: cost, SourceHash: hashJSON(sources), PromptHash: hashBytes([]byte(e.prompt)), + ImplementationSHA: strings.TrimSpace(string(sha)), CreatedAt: time.Now().UTC().Format(time.RFC3339), IncludePrereleases: allow, Force: force, Status: "prepared", + Reservation: "release-podcast-attempt-" + hashJSON(map[string]any{"repo": repository, "ids": ids, "mode": mode})} + m.GenerationHash = hashJSON(map[string]any{"sources": sources, "config": c, "prompt": e.prompt, "implementation": m.ImplementationSHA}) + return m, nil +} + +func (e *engine) writeFile(name string, data []byte) error { + if err := os.MkdirAll(e.output, 0750); err != nil { + return errors.New("cannot create output directory") + } + f, err := os.CreateTemp(e.output, ".release-podcast-*") + if err != nil { + return errors.New("cannot create output file") + } + defer os.Remove(f.Name()) + if _, err := f.Write(data); err != nil { + f.Close() + return errors.New("cannot write output") + } + if err := f.Close(); err != nil { + return errors.New("cannot close output") + } + if err := os.Rename(f.Name(), filepath.Join(e.output, name)); err != nil { + return errors.New("cannot install output") + } + return nil +} + +func (e *engine) writeJSON(name string, value any) error { + data, err := json.MarshalIndent(value, "", " ") + if err != nil { + return errors.New("cannot encode output") + } + return e.writeFile(name, append(data, '\n')) +} + +func (e *engine) readJSON(name string, target any) error { + data, err := os.ReadFile(filepath.Join(e.output, name)) + if err != nil { + return errors.New("required output checkpoint is unavailable") + } + if len(data) > 8<<20 || json.Unmarshal(data, target) != nil { + return errors.New("invalid output checkpoint") + } + return nil +} + +func (e *engine) saveSources(sources []source, m manifest) error { + if err := e.writeJSON("sources.json", sources); err != nil { + return err + } + return e.writeJSON("manifest.json", m) +} + +func (e *engine) reserveLocal(m manifest, force bool) (func(), error) { + if err := os.MkdirAll(e.output, 0750); err != nil { + return nil, errors.New("cannot create output directory") + } + lockPath := filepath.Join(e.output, ".release-podcast.lock") + lock, err := os.OpenFile(lockPath, os.O_CREATE|os.O_EXCL|os.O_WRONLY, 0600) + if err != nil { + return nil, errors.New("output directory is locked by another attempt; review a stale lock before removal") + } + lock.Close() + release := func() { _ = os.Remove(lockPath) } + fail := func(err error) (func(), error) { release(); return nil, err } + ledger := filepath.Join(e.output, ".attempts") + if err := os.MkdirAll(ledger, 0750); err != nil { + return fail(errors.New("cannot create local attempt ledger")) + } + attempt := filepath.Join(ledger, m.Reservation+".json") + if !force { + if err := e.checkGeneratedFiles(); err != nil { + return fail(err) + } + } + flags := os.O_CREATE | os.O_EXCL | os.O_WRONLY + if force { + flags = os.O_CREATE | os.O_TRUNC | os.O_WRONLY + } + f, err := os.OpenFile(attempt, flags, 0600) + if err != nil { + return fail(errors.New("paid attempt already reserved or ledger unavailable; explicit --force permits another charge")) + } + data, _ := json.Marshal(m) + _, writeErr := f.Write(data) + closeErr := f.Close() + if writeErr != nil || closeErr != nil { + return fail(errors.New("cannot persist attempt reservation; no paid request made")) + } + return release, nil +} + +func (e *engine) checkGeneratedFiles() error { + for _, name := range []string{"transcript.txt", "release-podcast.mp3", "evidence.json"} { + _, err := os.Lstat(filepath.Join(e.output, name)) + if err == nil { + return errors.New("generated files already exist; use a fresh output directory or explicit --force in a paid mode") + } + if !errors.Is(err, os.ErrNotExist) { + return errors.New("cannot inspect existing output files") + } + } + return nil +} + +func (e *engine) stage(ctx context.Context, kind string) (manifest, []source, error) { + var m manifest + var sources []source + if err := e.readJSON("manifest.json", &m); err != nil { + return m, nil, err + } + if err := e.readJSON("sources.json", &sources); err != nil { + return m, nil, err + } + if e.openAIKey == "" || m.Mode == "validate" { + return m, nil, errors.New("paid stage is not authorized or key is missing") + } + if (m.Origin != "local" && m.Origin != "github") || (os.Getenv("GITHUB_ACTIONS") == "true") != (m.Origin == "github") { + return m, nil, errors.New("checkpoint origin does not match the execution context") + } + if m.Origin == "github" { + if _, err := githubEvent(); err != nil { + return m, nil, err + } + if os.Getenv("AUDIO_ENABLED") != "true" { + return m, nil, errors.New("Actions paid generation is disabled") + } + if m.RunID != os.Getenv("GITHUB_RUN_ID") || m.RunAttempt != os.Getenv("GITHUB_RUN_ATTEMPT") { + return m, nil, errors.New("checkpoint belongs to another Actions attempt") + } + c, err := reviewedConfig(os.Getenv("AUDIO_TEXT_MODEL"), os.Getenv("AUDIO_TTS_MODEL"), os.Getenv("AUDIO_VOICE"), m.Mode) + if err != nil || c != m.Config { + return m, nil, errors.New("configuration changed after preparation") + } + } + if (kind == "text" && (m.TextRequests != 0 || m.Status != "prepared")) || (kind == "speech" && (m.SpeechRequests != 0 || m.Status != "script_validated" || m.Mode != "audio")) { + return m, nil, errors.New("stage already attempted or checkpoint not ready") + } + if hashJSON(sources) != m.SourceHash { + return m, nil, errors.New("source checkpoint changed") + } + if err := e.recheck(ctx, m, sources); err != nil { + return m, nil, err + } + return m, sources, nil +} + +func (e *engine) generateScript(ctx context.Context) error { + m, sources, err := e.stage(ctx, "text") + if err != nil { + return err + } + payload := e.textPayload(sources, m.Config) + if _, err := modeledCost(payload, m.Config, m.Mode); err != nil { + return err + } + m.TextRequests = 1 + m.Status = "script_requested" + if err := e.writeJSON("manifest.json", m); err != nil { + return err + } + data, ct, err := e.request(ctx, "https://api.openai.com/v1/responses", e.openAIKey, payload, 1<<20) + if err != nil { + return err + } + if ct != "application/json" { + return errors.New("text API returned an unexpected content type") + } + result, usage, err := parseResponse(data) + if err != nil { + return err + } + text, err := validateNarration(result, sources) + if err != nil { + return err + } + if len(strings.TrimSuffix(text, "\n")) > narrationByteLimit(m.Config) { + return errors.New("narration exceeds configured speech input limit; shorten optional highlights") + } + if err := e.writeFile("transcript.txt", []byte(text)); err != nil { + return err + } + if err := e.writeJSON("evidence.json", result); err != nil { + return err + } + m.Status = "script_validated" + m.ScriptHash = hashBytes([]byte(text)) + m.Words = len(strings.Fields(text)) + m.Characters = len([]rune(strings.TrimSuffix(text, "\n"))) + m.TextUsage = usage + return e.writeJSON("manifest.json", m) +} + +func (e *engine) mediaTools(ctx context.Context) error { + for _, tool := range []string{"ffprobe", "ffmpeg"} { + toolCtx, cancel := context.WithTimeout(ctx, 5*time.Second) + _, err := e.command(toolCtx, tool, "-version") + cancel() + if err != nil { + return fmt.Errorf("%s is required before paid audio generation", tool) + } + } + return nil +} + +func (e *engine) generateSpeech(ctx context.Context) error { + m, sources, err := e.stage(ctx, "speech") + if err != nil { + return err + } + text, err := os.ReadFile(filepath.Join(e.output, "transcript.txt")) + if err != nil { + return errors.New("script checkpoint unavailable") + } + var result narration + if err := e.readJSON("evidence.json", &result); err != nil { + return err + } + validated, err := validateNarration(result, sources) + if err != nil || validated != string(text) || hashBytes(text) != m.ScriptHash { + return errors.New("script checkpoint changed or is invalid") + } + if err := e.mediaTools(ctx); err != nil { + return err + } + input := strings.TrimSuffix(string(text), "\n") + payload := map[string]any{"model": m.Config.TTSModel, "voice": m.Config.Voice, "input": input, "response_format": "mp3", "speed": m.Config.Speed} + if m.Config.Instructions != "" { + if len(input)+len(m.Config.Instructions) > miniInputBytes { + return errors.New("mini TTS conservative input-token limit exceeded") + } + payload["instructions"] = m.Config.Instructions + } + m.SpeechRequests = 1 + m.Status = "speech_requested" + if err := e.writeJSON("manifest.json", m); err != nil { + return err + } + data, ct, err := e.request(ctx, "https://api.openai.com/v1/audio/speech", e.openAIKey, payload, 10<<20) + if err != nil { + return err + } + if (ct != "audio/mpeg" && ct != "audio/mp3" && ct != "application/octet-stream") || len(data) == 0 { + return errors.New("speech API did not return audio") + } + if err := e.writeFile("audio.tmp", data); err != nil { + return err + } + temporary := filepath.Join(e.output, "audio.tmp") + defer os.Remove(temporary) + mediaCtx, cancel := context.WithTimeout(ctx, 30*time.Second) + defer cancel() + probe, err := e.command(mediaCtx, "ffprobe", "-v", "error", "-f", "mp3", "-protocol_whitelist", "file,pipe", "-show_entries", "format=duration:stream=codec_name", "-of", "json", temporary) + if err != nil { + return errors.New("speech response cannot be inspected as MP3") + } + var info struct { + Format struct { + Duration string `json:"duration"` + } `json:"format"` + Streams []struct { + Codec string `json:"codec_name"` + } `json:"streams"` + } + if json.Unmarshal(probe, &info) != nil || len(info.Streams) == 0 { + return errors.New("invalid MP3 metadata") + } + duration, err := strconv.ParseFloat(info.Format.Duration, 64) + if err != nil || math.IsNaN(duration) || math.IsInf(duration, 0) || duration <= 0 { + return errors.New("invalid MP3 duration") + } + for _, stream := range info.Streams { + if stream.Codec != "mp3" { + return errors.New("speech response is not MP3") + } + } + if _, err := e.command(mediaCtx, "ffmpeg", "-v", "error", "-xerror", "-f", "mp3", "-protocol_whitelist", "file,pipe", "-i", temporary, "-f", "null", "-"); err != nil { + return errors.New("MP3 failed complete decoding") + } + if err := os.Rename(temporary, filepath.Join(e.output, "release-podcast.mp3")); err != nil { + return errors.New("cannot install validated MP3") + } + m.Status = "audio_validated" + m.Duration = duration + m.DurationNeedsReview = duration < 105 || duration > 145 + m.AudioHash = hashBytes(data) + return e.writeJSON("manifest.json", m) +} + +func strictDecode(data []byte, target any) error { + decoder := json.NewDecoder(bytes.NewReader(data)) + decoder.DisallowUnknownFields() + if err := decoder.Decode(target); err != nil { + return errors.New("invalid structured narration schema") + } + var trailing any + if err := decoder.Decode(&trailing); err != io.EOF { + return errors.New("unexpected data after narration") + } + return nil +} diff --git a/release/podcast/podcast_test.go b/release/podcast/podcast_test.go new file mode 100644 index 000000000..dfe05672e --- /dev/null +++ b/release/podcast/podcast_test.go @@ -0,0 +1,852 @@ +package main + +import ( + "bytes" + "context" + "encoding/json" + "errors" + "io" + "net/http" + "os" + "os/exec" + "path/filepath" + "strconv" + "strings" + "testing" +) + +type roundTripFunc func(*http.Request) (*http.Response, error) + +func (f roundTripFunc) RoundTrip(r *http.Request) (*http.Response, error) { return f(r) } + +func fixtures(t *testing.T) ([]releaseRecord, []source) { + t.Helper() + data, err := os.ReadFile("testdata/releases.json") + if err != nil { + t.Fatal(err) + } + var records []releaseRecord + if err := json.Unmarshal(data, &records); err != nil { + t.Fatal(err) + } + var sources []source + for _, r := range records { + s, err := normalizeRelease(r, false) + if err != nil { + t.Fatal(err) + } + sources = append(sources, s) + } + return records, sources +} + +func exampleNarration(t *testing.T, sources []source) narration { + t.Helper() + texts := []struct { + index int + text, quote string + }{ + {0, "Version 0.64.0 introduced experimental Jellyfin music support, which must be explicitly enabled.", "Enable it with `Jellyfin.Enabled = true`."}, + {0, "Back up your database before upgrading because internal IDs change; clients may need to resync cached IDs.", "back up your database before upgrading"}, + {0, "Plugin authors must migrate to the host HTTP service and review restrictions on private or loopback network addresses.", "Plugins must use the host HTTP service instead"}, + {0, "Shares now belong to their creator, and admins cannot create them for another user.", "Shares are always owned by the user who creates them."}, + {0, "Negative configuration durations are rejected at startup, and unknown options produce warnings.", "Negative values are rejected at startup."}, + {0, "Security fixes protect library access and plugin networking, alongside improvements to artwork, sorting and playlist imports.", "This release fixes several reported vulnerabilities."}, + {0, "Database restore also avoids wiping existing data when the backup file is missing.", "wiping the database when the backup file does not exist"}, + {1, "Version 0.64.1 is a security release fixing five vulnerabilities; upgrade as soon as practical.", "This is a security release. It fixes five vulnerabilities"}, + {1, "Jellyfin client compatibility improves, and Quick Connect makes signing in easier.", "Add Quick Connect sign-in."}, + {1, "Local discovery is opt-in, and Docker users need host networking for discovery broadcasts.", "Docker users need host networking for the UDP broadcast to reach the container."}, + {1, "Smart playlists can reference another playlist by path, while the interface follows your selected language for dates.", "Smart playlists can reference another playlist by path"}, + {2, "Version 0.64.2 fixes scan failures and database lock contention on slow storage.", "Fix `database is locked` errors, UI freezes and failed scans on slow storage."}, + {2, "It also fixes scans on 32-bit builds with invalid track metadata.", "32-bit builds (armv5/6/7, 386)"}, + {2, "Security fixes sanitize download names and prevent an admin password from reaching logs.", "Stop writing the admin password to the log"}, + } + r := narration{Cautions: []coverage{}} + for _, item := range texts { + if !strings.Contains(sources[item.index].Body, item.quote) { + t.Fatalf("fixture quote absent: %s", item.quote) + } + r.Sentences = append(r.Sentences, sentence{Text: item.text, SourceID: sources[item.index].SourceID, Excerpt: item.quote}) + } + for _, c := range cautions(sources) { + index := 0 + for i, s := range r.Sentences { + if s.SourceID == c.SourceID { + index = i + break + } + } + r.Cautions = append(r.Cautions, coverage{CautionID: c.ID, SentenceIndex: &index}) + } + return r +} + +func response(status int, ct string, data []byte) *http.Response { + return &http.Response{StatusCode: status, Header: http.Header{"Content-Type": []string{ct}}, Body: io.NopCloser(bytes.NewReader(data))} +} + +func testEngine(t *testing.T) (*engine, []releaseRecord, []source) { + t.Helper() + // Local fixtures must not inherit the hosting CI runner's Actions context. + // Actions-specific tests establish their own context with actionsEnvironment. + t.Setenv("GITHUB_ACTIONS", "false") + t.Setenv("GH_TOKEN", "") + t.Setenv("OPENAI_API_KEY", "") + e, err := newEngine(t.TempDir()) + if err != nil { + t.Fatal(err) + } + e.openAIKey = "offline-test-key" + records, sources := fixtures(t) + e.command = func(_ context.Context, name string, args ...string) ([]byte, error) { + if name == "git" { + return []byte("test-commit"), nil + } + if len(args) == 1 && args[0] == "-version" { + return nil, nil + } + if name == "ffprobe" { + return []byte(`{"format":{"duration":"120.5"},"streams":[{"codec_name":"mp3"}]}`), nil + } + if name == "ffmpeg" { + return nil, nil + } + return nil, errors.New("unexpected command") + } + e.client.Transport = roundTripFunc(func(req *http.Request) (*http.Response, error) { + if req.URL.Host == "api.github.com" { + if req.URL.Path == "/repos/"+repository+"/releases" { + data, _ := json.Marshal(records) + return response(200, "application/json", data), nil + } + if strings.Contains(req.URL.Path, "/actions/artifacts") { + return response(200, "application/json", []byte(`{"artifacts":[]}`)), nil + } + for _, record := range records { + if req.URL.Path == "/repos/"+repository+"/releases/tags/"+record.Tag || req.URL.Path == "/repos/"+repository+"/releases/"+strconv.FormatInt(record.ID, 10) { + data, _ := json.Marshal(record) + return response(200, "application/json", data), nil + } + } + } + if req.URL.Host == "api.openai.com" { + if req.URL.Path == "/v1/responses" { + data, _ := json.Marshal(exampleNarration(t, sources)) + body, _ := json.Marshal(map[string]any{"status": "completed", "usage": map[string]int{"input_tokens": 1000}, "output": []any{map[string]any{"type": "message", "content": []any{map[string]string{"type": "output_text", "text": string(data)}}}}}) + return response(200, "application/json", body), nil + } + if req.URL.Path == "/v1/audio/speech" { + return response(200, "audio/mpeg", []byte("mock-mp3-data")), nil + } + } + t.Fatalf("unmocked network request: %s", req.URL.Host+req.URL.Path) + return nil, errors.New("unmocked request") + }) + return e, records, sources +} + +func audioOptions() options { + return options{from: "v0.64.0", to: "v0.64.2", mode: "audio", textModel: "gpt-6-luna", ttsModel: "gpt-4o-mini-tts-2025-12-15", voice: "onyx", allowPaid: true} +} + +func TestVersionSelection(t *testing.T) { + for _, raw := range []string{"", "v0.64.0,", "v0.64.0,0.64.0", "v1.0.0,v2.0.0,v3.0.0,v4.0.0", "$(touch secret)", "../../foo", "v1.0.0\nmalicious", "v01.2.3", "v1.2.3١", "v99999999999.0.0"} { + t.Run(raw, func(t *testing.T) { + if _, err := normalizeTag(raw); err == nil { + t.Fatal("unsafe tags accepted") + } + }) + } + tag, err := normalizeTag("0.64.2") + if err != nil || tag != "v0.64.2" { + t.Fatal(tag, err) + } +} + +func TestSingleReleaseSelection(t *testing.T) { + e, records, _ := testEngine(t) + original := e.client.Transport + e.client.Transport = roundTripFunc(func(req *http.Request) (*http.Response, error) { + if req.URL.Path != "/repos/"+repository+"/releases/tags/v0.64.2" { + t.Fatal("single release must use its exact tag endpoint", req.URL) + } + return original.RoundTrip(req) + }) + for _, from := range []string{"0.64.2", "v0.64.2"} { + sources, err := e.resolveLocal(t.Context(), options{from: from}) + if err != nil || len(sources) != 1 || sources[0].Tag != "v0.64.2" { + t.Fatal(sources, err) + } + } + if _, err := parseOptions([]string{"--tags", "v0.64.0,v0.64.1"}, io.Discard); err == nil { + t.Fatal("removed --tags flag remains accepted") + } + r := records[0] + r.Tag += "-rc.1" + r.Prerelease = true + data, _ := json.Marshal(r) + e.client.Transport = roundTripFunc(func(*http.Request) (*http.Response, error) { return response(200, "application/json", data), nil }) + if _, err := e.resolveLocal(t.Context(), options{from: r.Tag}); err == nil { + t.Fatal("single prerelease accepted without opt-in") + } + if sources, err := e.resolveLocal(t.Context(), options{from: r.Tag, includePrereleases: true}); err != nil || len(sources) != 1 { + t.Fatal(sources, err) + } +} + +func TestReleaseEligibility(t *testing.T) { + records, _ := fixtures(t) + cases := map[string]func(*releaseRecord){"draft": func(r *releaseRecord) { r.Draft = true }, "prerelease": func(r *releaseRecord) { r.Prerelease = true }, "empty": func(r *releaseRecord) { r.Body = " " }, "unpublished": func(r *releaseRecord) { r.PublishedAt = "" }, "id": func(r *releaseRecord) { r.ID = 0 }, "oversized": func(r *releaseRecord) { r.Body = strings.Repeat("x", maxSourceBytes+1) }, "multiple tags": func(r *releaseRecord) { r.Tag = "v0.64.0,v0.64.1" }} + for name, mutate := range cases { + t.Run(name, func(t *testing.T) { + r := records[0] + mutate(&r) + if _, err := normalizeRelease(r, false); err == nil { + t.Fatal("ineligible release accepted") + } + }) + } + r := records[0] + r.Prerelease = true + r.Tag = "v0.64.0-rc.1" + if _, err := normalizeRelease(r, true); err != nil { + t.Fatal(err) + } +} + +func TestInclusiveRangeSelection(t *testing.T) { + e, _, _ := testEngine(t) + sources, err := e.resolveLocal(t.Context(), options{from: "0.64.0", to: "v0.64.2"}) + if err != nil || len(sources) != 3 || sources[0].Tag != "v0.64.0" || sources[2].Tag != "v0.64.2" { + t.Fatal(sources, err) + } + for _, o := range []options{{from: "v0.64.2", to: "v0.64.0"}, {from: "v0.63.0", to: "v0.64.2"}, {}, {to: "v0.64.2"}, {from: "v0.64.0,v0.64.1"}, {from: "v0.64.0-rc.1", to: "v0.64.2"}} { + if _, err := e.resolveLocal(t.Context(), o); err == nil { + t.Fatal("invalid range accepted", o) + } + } +} + +func TestSemVerPrereleasePrecedence(t *testing.T) { + ordered := []string{"v1.0.0-alpha", "v1.0.0-alpha.1", "v1.0.0-alpha.beta", "v1.0.0-beta", "v1.0.0-beta.2", "v1.0.0-beta.11", "v1.0.0-rc.1", "v1.0.0"} + for i, a := range ordered { + for j, b := range ordered { + order := compareVersion(a, b) + if (i < j && order >= 0) || (i == j && order != 0) || (i > j && order <= 0) { + t.Fatalf("incorrect precedence for %s and %s: %d", a, b, order) + } + } + } + for _, pair := range [][2]string{{"v1.0.0-rc.2", "v1.0.0-rc.10"}, {"v1.0.0-99999999999999999999", "v1.0.0-100000000000000000000"}, {"v1.0.0-9", "v1.0.0-alpha"}, {"v1.0.0-alpha.beta", "v1.0.0-alpha-beta"}, {"v1.0.0", "v1.0.1-alpha"}} { + if _, err := version(pair[0]); err != nil { + t.Fatal(err) + } + if compareVersion(pair[0], pair[1]) >= 0 { + t.Fatal("incorrect numeric/identifier precedence", pair) + } + } + for _, tag := range []string{"v1.0.0-01", "v1.0.0-rc.01", "v1.0.0-rc..1", "v1.0.0-"} { + if _, err := normalizeTag(tag); err == nil { + t.Fatal("invalid SemVer prerelease accepted", tag) + } + } +} + +func TestRangePrereleaseBoundaries(t *testing.T) { + e, records, _ := testEngine(t) + selected := []releaseRecord{records[1], records[0]} + for i := range 2 { + r := records[i] + r.ID += 100 + r.Tag += "-rc.1" + r.Prerelease = true + selected = append(selected, r) + } + data, _ := json.Marshal(selected) + e.client.Transport = roundTripFunc(func(*http.Request) (*http.Response, error) { return response(200, "application/json", data), nil }) + sources, err := e.resolveLocal(t.Context(), options{from: "v0.64.0", to: "v0.64.1", includePrereleases: true}) + if err != nil || len(sources) != 3 || sources[0].Tag != "v0.64.0" || sources[1].Tag != "v0.64.1-rc.1" || sources[2].Tag != "v0.64.1" { + t.Fatal("prerelease at lower bound must be excluded, upper-bound RC included", sources, err) + } + sources, err = e.resolveLocal(t.Context(), options{from: "v0.64.0", to: "v0.64.1"}) + if err != nil || len(sources) != 2 { + t.Fatal("prerelease opt-in was bypassed", sources, err) + } +} + +func TestRangeDiscardsDraftsAndFailsClosedAtLimit(t *testing.T) { + e, records, _ := testEngine(t) + draft := records[0] + draft.ID = 999 + draft.Draft = true + draft.Tag = "v0.64.0-rc.1" + draft.Body = "private advisory never sent" + data, _ := json.Marshal(append(records, draft)) + e.client.Transport = roundTripFunc(func(*http.Request) (*http.Response, error) { return response(200, "application/json", data), nil }) + sources, err := e.resolveLocal(t.Context(), options{from: "v0.64.0", to: "v0.64.2", includePrereleases: true}) + if err != nil || len(sources) != 3 { + t.Fatal(sources, err) + } + page := make([]releaseRecord, 100) + for i := range page { + page[i] = releaseRecord{Draft: true} + } + data, _ = json.Marshal(page) + calls := 0 + e.client.Transport = roundTripFunc(func(*http.Request) (*http.Response, error) { + calls++ + return response(200, "application/json", data), nil + }) + if _, err := e.resolveLocal(t.Context(), options{from: "v0.64.0", to: "v0.64.2"}); err == nil || calls != 10 { + t.Fatal("unbounded/incomplete range accepted", calls, err) + } +} + +func TestConfigurationAndCostLimits(t *testing.T) { + e, _, sources := testEngine(t) + cfg, err := reviewedConfig("gpt-6-luna", "gpt-4o-mini-tts-2025-12-15", "cedar", "audio") + if err != nil { + t.Fatal(err) + } + payload := e.textPayload(sources, cfg) + cost, err := modeledCost(payload, cfg, "audio") + if err != nil || cost == nil || *cost > maxCostUSD { + t.Fatal(cost, err) + } + if payload["store"] != false || payload["tools"] != nil || payload["reasoning"] == nil { + t.Fatal("unsafe text configuration") + } + if _, err := modeledCost(strings.Repeat("x", maxPromptBytes+1), cfg, "audio"); err == nil { + t.Fatal("oversized prompt accepted") + } + hd, err := reviewedConfig("gpt-4.1-mini-2025-04-14", "tts-1-hd", "onyx", "audio") + if err != nil { + t.Fatal(err) + } + if _, err := modeledCost(strings.Repeat("x", 60000), hd, "audio"); err == nil { + t.Fatal("modeled dollar ceiling ignored") + } + for _, values := range [][3]string{{"", "", ""}, {"unreviewed", "tts-1", "onyx"}, {"gpt-6-luna", "unreviewed", "onyx"}, {"gpt-6-luna", "tts-1", "cedar"}, {"gpt-6-luna", "tts-1", "echo,fable"}} { + if _, err := reviewedConfig(values[0], values[1], values[2], "audio"); err == nil { + t.Fatal("invalid configuration accepted", values) + } + } + if _, err := reviewedConfig("", "", "", "validate"); err != nil { + t.Fatal(err) + } + legacy, err := reviewedConfig("gpt-6-luna", "tts-1", "onyx", "audio") + if err != nil || legacy.Instructions != "" { + t.Fatal(legacy, err) + } +} + +func TestCLIDryRunAndPaidGuards(t *testing.T) { + t.Setenv("AUDIO_TEXT_MODEL", "gpt-6-luna") + o, err := parseOptions([]string{"--from", "0.64.2", "--text-model", "gpt-4.1-mini-2025-04-14", "--mode", "audio", "--dry-run", "--output", "out"}, io.Discard) + if err != nil || o.mode != "validate" || o.textModel != "gpt-4.1-mini-2025-04-14" || o.output != "out" { + t.Fatal(o, err) + } + if _, err := parseOptions([]string{"--openai-key", "never-pass-a-key"}, io.Discard); err == nil { + t.Fatal("key argument accepted") + } + e, _, _ := testEngine(t) + paid := audioOptions() + paid.allowPaid = false + if err := e.runLocal(t.Context(), paid, io.Discard); err == nil { + t.Fatal("paid mode needs explicit flag") + } + paid.allowPaid = true + e.openAIKey = "" + if err := e.runLocal(t.Context(), paid, io.Discard); err == nil { + t.Fatal("paid mode needs local environment key") + } +} + +func TestLocalValidationNeverContactsOpenAI(t *testing.T) { + e, _, _ := testEngine(t) + original := e.client.Transport + e.client.Transport = roundTripFunc(func(req *http.Request) (*http.Response, error) { + if req.URL.Host == "api.openai.com" { + t.Fatal("validation contacted OpenAI") + } + return original.RoundTrip(req) + }) + if err := e.runLocal(t.Context(), options{from: "v0.64.0", to: "v0.64.2", mode: "validate"}, io.Discard); err != nil { + t.Fatal(err) + } + var m manifest + if err := e.readJSON("manifest.json", &m); err != nil || m.TextRequests != 0 || m.SpeechRequests != 0 || m.Origin != "local" { + t.Fatal(m, err) + } +} + +func TestLocalPaidPipelineAndDuplicateProtection(t *testing.T) { + e, _, _ := testEngine(t) + if err := e.runLocal(t.Context(), audioOptions(), io.Discard); err != nil { + t.Fatal(err) + } + var m manifest + if err := e.readJSON("manifest.json", &m); err != nil { + t.Fatal(err) + } + if m.Status != "audio_validated" || m.TextRequests != 1 || m.SpeechRequests != 1 || m.AudioHash == "" || m.ScriptHash == "" || m.Duration != 120.5 { + t.Fatal(m) + } + for _, name := range []string{"release-podcast.mp3", "transcript.txt", "evidence.json", "sources.json", "manifest.json"} { + if _, err := os.Stat(filepath.Join(e.output, name)); err != nil { + t.Fatal(err) + } + } + if err := e.runLocal(t.Context(), audioOptions(), io.Discard); err == nil { + t.Fatal("duplicate attempt permitted") + } + o := audioOptions() + o.force = true + if err := e.runLocal(t.Context(), o, io.Discard); err != nil { + t.Fatal(err) + } +} + +func TestLocalLedgerPersistsFailuresAndLocksConcurrentAttempts(t *testing.T) { + e, _, sources := testEngine(t) + cfg, _ := reviewedConfig("gpt-6-luna", "tts-1", "onyx", "audio") + m, err := e.prepare(t.Context(), sources, cfg, "audio", "local", false, false) + if err != nil { + t.Fatal(err) + } + release, err := e.reserveLocal(m, false) + if err != nil { + t.Fatal(err) + } + if _, err := e.reserveLocal(m, true); err == nil { + t.Fatal("force bypassed active lock") + } + release() + if _, err := e.reserveLocal(m, false); err == nil { + t.Fatal("reserved failed attempt retried") + } + release, err = e.reserveLocal(m, true) + if err != nil { + t.Fatal(err) + } + release() +} + +func TestEvidenceAndConsequentialOmissions(t *testing.T) { + _, sources := fixtures(t) + if _, err := validateNarration(exampleNarration(t, sources), sources); err != nil { + t.Fatal(err) + } + for _, term := range []string{"Back up", "resync", "experimental", "host HTTP", "opt-in", "host networking", "32-bit", "slow storage"} { + t.Run(term, func(t *testing.T) { + r := exampleNarration(t, sources) + for i := range r.Sentences { + r.Sentences[i].Text = strings.ReplaceAll(r.Sentences[i].Text, term, "some detail") + } + if _, err := validateNarration(r, sources); err == nil { + t.Fatal("consequential qualifier omitted") + } + }) + } + for name, mutate := range map[string]func(*narration){"source": func(r *narration) { r.Sentences[0].SourceID = "unknown" }, "excerpt": func(r *narration) { r.Sentences[0].Excerpt = "fabricated unsupported evidence" }, "coverage": func(r *narration) { r.Cautions = r.Cautions[1:] }, "index": func(r *narration) { n := 999; r.Cautions[0].SentenceIndex = &n }, "duplicate": func(r *narration) { r.Cautions = append(r.Cautions, r.Cautions[0]) }} { + t.Run(name, func(t *testing.T) { + r := exampleNarration(t, sources) + mutate(&r) + if _, err := validateNarration(r, sources); err == nil { + t.Fatal("invalid evidence accepted") + } + }) + } +} + +func TestUnsafeModelOutputAndSourceMarkup(t *testing.T) { + _, sources := fixtures(t) + for _, suffix := range []string{" https://evil.example", " `shell`", " $(cat secret)", "