330 lines
11 KiB
Go
330 lines
11 KiB
Go
package credits
|
|
|
|
import (
|
|
"bytes"
|
|
"context"
|
|
"errors"
|
|
"fmt"
|
|
"io"
|
|
"os/exec"
|
|
"strconv"
|
|
"strings"
|
|
"time"
|
|
)
|
|
|
|
// FFmpeg, used surgically and never as a transcoder.
|
|
//
|
|
// The rules this file exists to keep are the ones that make the difference between reading a
|
|
// couple of megabytes and reading a whole film: seek before opening the input so the decoder
|
|
// starts at the credits rather than reading its way there, bound the read with -t, throw away
|
|
// audio and subtitles, downscale to a thumbnail, drop to a frame every few seconds, and take
|
|
// the result as raw grayscale on stdout. Nothing is ever written to disk — no JPEGs, no
|
|
// temporary transcode, no scratch file. The frames exist only as bytes in a buffer that is
|
|
// reused between passes.
|
|
//
|
|
// The gateway has no filesystem access to the media (docker-compose mounts no media share),
|
|
// so the input is Emby's own stream route over HTTP. That is not a compromise: -ss before -i
|
|
// makes ffmpeg issue a ranged request, so the bytes that cross the network are the bytes of
|
|
// the window and not of the file.
|
|
|
|
const (
|
|
// The sampling grid. Small enough that a frame is fourteen kilobytes and the statistics
|
|
// are computed in a few microseconds, large enough that a credits roll still reads as
|
|
// structured rather than as noise.
|
|
sampleWidth = 160
|
|
sampleHeight = 90
|
|
frameBytes = sampleWidth * sampleHeight
|
|
|
|
// coarseInterval is the first pass: one frame every four seconds, which is enough to
|
|
// find a transition to within a few seconds while sampling a two-minute window in about
|
|
// thirty frames.
|
|
coarseInterval = 4 * time.Second
|
|
// fineInterval is the second pass, run only over the span the first pass pointed at.
|
|
fineInterval = 750 * time.Millisecond
|
|
// fineSpan is how much of the file either side of the coarse estimate the fine pass
|
|
// covers.
|
|
fineSpan = 30 * time.Second
|
|
|
|
// maxFrames is a hard ceiling on one pass. It bounds memory (frames are held only one at
|
|
// a time, but the statistics slice is not) and, more importantly, bounds the damage a
|
|
// mis-computed window can do: without it a bad runtime could turn a tail scan into a
|
|
// full decode.
|
|
maxFrames = 400
|
|
)
|
|
|
|
// ErrNoFFmpeg means the binary is absent. Reported distinctly so the service can stand the
|
|
// visual detector down and run on behaviour alone rather than logging a decoder failure per
|
|
// candidate for the life of the container.
|
|
var ErrNoFFmpeg = errors.New("credits: ffmpeg is not available")
|
|
|
|
// Sampler decodes a span of a file into frame statistics.
|
|
type Sampler struct {
|
|
// Binary is the ffmpeg executable. Configurable because a NAS may carry it somewhere
|
|
// other than the path.
|
|
Binary string
|
|
// Timeout bounds one pass. A decoder that hangs on a malformed file must not hold the
|
|
// single worker for ever.
|
|
Timeout time.Duration
|
|
}
|
|
|
|
// frameStats is one sampled frame reduced to the handful of cheap properties a credits
|
|
// transition shows up in. Deliberately not the frame: nothing downstream needs the picture,
|
|
// and keeping four hundred thumbnails would be most of the package's memory budget.
|
|
type frameStats struct {
|
|
PositionMs int64
|
|
// Mean luminance, 0-1. Credits are dark.
|
|
Mean float64
|
|
// Variance of luminance, 0-1 scaled. A credits roll is mostly flat background with thin
|
|
// text, so its variance is low and, more usefully, *stable*.
|
|
Variance float64
|
|
// DarkFraction is the proportion of pixels below the dark threshold.
|
|
DarkFraction float64
|
|
// EdgeDensity approximates how much fine detail there is, which is what separates a
|
|
// credits roll from a dark night scene: text has edges, darkness does not.
|
|
EdgeDensity float64
|
|
// Diff is the mean absolute difference from the previous sampled frame. Scrolling text
|
|
// changes steadily; a held black frame does not.
|
|
Diff float64
|
|
}
|
|
|
|
// Available reports whether the decoder can be used at all.
|
|
func (s *Sampler) Available() bool {
|
|
_, err := exec.LookPath(s.binary())
|
|
return err == nil
|
|
}
|
|
|
|
func (s *Sampler) binary() string {
|
|
if strings.TrimSpace(s.Binary) != "" {
|
|
return s.Binary
|
|
}
|
|
return "ffmpeg"
|
|
}
|
|
|
|
// Sample decodes one span and returns its frame statistics.
|
|
//
|
|
// The byte count it reports is an estimate — ffmpeg does not report how much of its input it
|
|
// read, and adding a proxy to find out would cost more than the number is worth. It is
|
|
// derived from the span and the file's bitrate, which is accurate enough for the question it
|
|
// answers: whether this subsystem is reading a couple of minutes or the whole file.
|
|
func (s *Sampler) Sample(
|
|
ctx context.Context, url string, from, to time.Duration, interval time.Duration,
|
|
) ([]frameStats, error) {
|
|
if to <= from || url == "" {
|
|
return nil, nil
|
|
}
|
|
if interval <= 0 {
|
|
interval = coarseInterval
|
|
}
|
|
timeout := s.Timeout
|
|
if timeout <= 0 {
|
|
timeout = 60 * time.Second
|
|
}
|
|
ctx, cancel := context.WithTimeout(ctx, timeout)
|
|
defer cancel()
|
|
|
|
stats, err := s.samplePass(ctx, url, from, to, interval, false)
|
|
if err == nil || !decoderCrashed(err) || ctx.Err() != nil {
|
|
return stats, err
|
|
}
|
|
// A decoder crash is local to the ffmpeg process, not evidence that the media is
|
|
// unreadable. Retry once with conservative decoder settings: single-threaded decoding
|
|
// avoids the most common native-code race, while corrupt packets are discarded rather
|
|
// than handed back through the failing path. Ordinary HTTP and authentication failures
|
|
// are never retried here.
|
|
stats, retryErr := s.samplePass(ctx, url, from, to, interval, true)
|
|
if retryErr != nil {
|
|
return stats, fmt.Errorf("credits: conservative ffmpeg retry: %w", retryErr)
|
|
}
|
|
return stats, nil
|
|
}
|
|
|
|
func (s *Sampler) samplePass(
|
|
ctx context.Context, url string, from, to time.Duration, interval time.Duration,
|
|
conservative bool,
|
|
) ([]frameStats, error) {
|
|
|
|
// -ss ahead of -i is the whole optimisation: it seeks in the container before opening a
|
|
// decoder, so the input starts near the credits. Behind -i it would decode from zero and
|
|
// discard, which is the full read this package exists to avoid.
|
|
args := []string{
|
|
"-hide_banner", "-loglevel", "error", "-nostdin",
|
|
"-ss", formatSeconds(from),
|
|
}
|
|
if conservative {
|
|
args = append(args,
|
|
"-threads", "1",
|
|
"-fflags", "+discardcorrupt",
|
|
"-err_detect", "ignore_err",
|
|
)
|
|
}
|
|
args = append(args,
|
|
"-i", url,
|
|
"-t", formatSeconds(to-from),
|
|
"-an", "-sn", "-dn",
|
|
"-vf", fmt.Sprintf("fps=%s,scale=%d:%d,format=gray",
|
|
formatRate(interval), sampleWidth, sampleHeight),
|
|
"-frames:v", strconv.Itoa(maxFrames),
|
|
"-f", "rawvideo", "-pix_fmt", "gray",
|
|
"pipe:1",
|
|
)
|
|
|
|
cmd := exec.CommandContext(ctx, s.binary(), args...)
|
|
// Cancel and WaitDelay together are what stop an orphan. CommandContext's default is to
|
|
// send Kill and then wait for the pipes to close, which a stuck HTTP read can hold open
|
|
// indefinitely; WaitDelay puts a bound on that and closes the descriptors itself.
|
|
cmd.Cancel = func() error { return cmd.Process.Kill() }
|
|
cmd.WaitDelay = 5 * time.Second
|
|
|
|
stdout, err := cmd.StdoutPipe()
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
var stderr bytes.Buffer
|
|
cmd.Stderr = &stderr
|
|
|
|
if err := cmd.Start(); err != nil {
|
|
if errors.Is(err, exec.ErrNotFound) {
|
|
return nil, ErrNoFFmpeg
|
|
}
|
|
return nil, fmt.Errorf("credits: start ffmpeg: %w", err)
|
|
}
|
|
|
|
stats, readErr := readFrames(stdout, from, interval)
|
|
// Drain whatever is left so ffmpeg is never blocked writing into a pipe nobody is
|
|
// reading, which is how a "finished" scan comes to sit in Wait for its full timeout.
|
|
_, _ = io.Copy(io.Discard, stdout)
|
|
waitErr := cmd.Wait()
|
|
|
|
if readErr != nil {
|
|
return nil, readErr
|
|
}
|
|
if waitErr != nil && len(stats) == 0 {
|
|
// A pass that produced frames and then failed is a truncated read, not a failure:
|
|
// the statistics that arrived are still usable. One that produced nothing is a
|
|
// genuine problem worth reporting with whatever ffmpeg said about it.
|
|
if ctx.Err() != nil {
|
|
return nil, ctx.Err()
|
|
}
|
|
return nil, fmt.Errorf("credits: ffmpeg: %w: %s",
|
|
waitErr, strings.TrimSpace(truncate(stderr.String(), 300)))
|
|
}
|
|
return stats, nil
|
|
}
|
|
|
|
func decoderCrashed(err error) bool {
|
|
if err == nil {
|
|
return false
|
|
}
|
|
message := strings.ToLower(err.Error())
|
|
for _, signature := range []string{
|
|
"segmentation fault",
|
|
"signal: aborted",
|
|
"signal: bus error",
|
|
"access violation",
|
|
} {
|
|
if strings.Contains(message, signature) {
|
|
return true
|
|
}
|
|
}
|
|
return false
|
|
}
|
|
|
|
// readFrames pulls fixed-size grayscale frames off the pipe and reduces each one as it
|
|
// arrives. The frame buffer is allocated once and reused, so a four-hundred-frame pass
|
|
// allocates fourteen kilobytes rather than five and a half megabytes.
|
|
func readFrames(reader io.Reader, from, interval time.Duration) ([]frameStats, error) {
|
|
frame := make([]byte, frameBytes)
|
|
stats := make([]frameStats, 0, 64)
|
|
var previous []byte
|
|
previousBuffer := make([]byte, frameBytes)
|
|
|
|
for index := 0; index < maxFrames; index++ {
|
|
if _, err := io.ReadFull(reader, frame); err != nil {
|
|
if errors.Is(err, io.EOF) || errors.Is(err, io.ErrUnexpectedEOF) {
|
|
break
|
|
}
|
|
return stats, err
|
|
}
|
|
position := from + time.Duration(index)*interval
|
|
stats = append(stats, analyseFrame(frame, previous, position))
|
|
copy(previousBuffer, frame)
|
|
previous = previousBuffer
|
|
}
|
|
return stats, nil
|
|
}
|
|
|
|
// analyseFrame reduces one thumbnail to its statistics in a single pass over the pixels,
|
|
// which at 14,400 bytes is a few microseconds. Nothing here allocates.
|
|
func analyseFrame(frame, previous []byte, position time.Duration) frameStats {
|
|
const darkThreshold = 48 // out of 255
|
|
|
|
var sum, sumSquares, dark, diff float64
|
|
for index, pixel := range frame {
|
|
value := float64(pixel)
|
|
sum += value
|
|
sumSquares += value * value
|
|
if pixel < darkThreshold {
|
|
dark++
|
|
}
|
|
if previous != nil {
|
|
delta := value - float64(previous[index])
|
|
if delta < 0 {
|
|
delta = -delta
|
|
}
|
|
diff += delta
|
|
}
|
|
}
|
|
count := float64(len(frame))
|
|
mean := sum / count
|
|
variance := sumSquares/count - mean*mean
|
|
if variance < 0 {
|
|
variance = 0
|
|
}
|
|
|
|
// Edge density as a horizontal gradient: credits are text, and text on a flat background
|
|
// is almost entirely horizontal transitions. A proper Sobel would cost a second pass and
|
|
// a second buffer for a distinction nothing downstream makes.
|
|
var edges float64
|
|
for row := 0; row < sampleHeight; row++ {
|
|
base := row * sampleWidth
|
|
for column := 1; column < sampleWidth; column++ {
|
|
delta := float64(frame[base+column]) - float64(frame[base+column-1])
|
|
if delta < 0 {
|
|
delta = -delta
|
|
}
|
|
if delta > 24 {
|
|
edges++
|
|
}
|
|
}
|
|
}
|
|
|
|
return frameStats{
|
|
PositionMs: position.Milliseconds(),
|
|
Mean: mean / 255,
|
|
Variance: variance / (255 * 255),
|
|
DarkFraction: dark / count,
|
|
EdgeDensity: edges / count,
|
|
Diff: diff / count / 255,
|
|
}
|
|
}
|
|
|
|
// formatSeconds writes a duration the way ffmpeg's -ss wants it, with millisecond precision
|
|
// and no unit suffix.
|
|
func formatSeconds(value time.Duration) string {
|
|
return strconv.FormatFloat(value.Seconds(), 'f', 3, 64)
|
|
}
|
|
|
|
// formatRate turns a sampling interval into an fps filter argument. Expressed as a fraction
|
|
// rather than a decimal because one frame every four seconds is 1/4 exactly and 0.25 is not,
|
|
// on a filter that accumulates rounding across a long window.
|
|
func formatRate(interval time.Duration) string {
|
|
return fmt.Sprintf("1000/%d", interval.Milliseconds())
|
|
}
|
|
|
|
func truncate(value string, limit int) string {
|
|
if len(value) <= limit {
|
|
return value
|
|
}
|
|
return value[:limit] + "…"
|
|
}
|