  /*=============================================================================
  *
  * UDP -> HLS IPTV Streaming Engine
  * ---
  * No-FFmpeg | All-Player Compatible | High Performance IPTV Engine
  *
  * Base Version  : V35
  * Engine		   : RStreaming Engine 3.0
  * Created Date   : 25-06-2025
  *
  * Developed By   : KALIDASS R
  * Company 		: Ridsys IPTV Solutions
  *
  * Copyright (c) 2025 Ridsys Solutions
  * All Rights Reserved.
  *
  *===========================================================================

  BUILD:
    gcc -O2 -pthread -o udp_hls udp_hls.c \
        -lssl -lcrypto -lmpg123 -lavcodec -lavutil -lswresample -lm
    # Install: apt-get install libopenh264-dev  OR  libx264-dev
    # OpenH264 is preferred; libx264 used as fallback

  OPTS:
    audio_transcode=1   MP2/MP3 -> AAC-LC  (~5% CPU, required for ExoPlayer/VLC)
    fix_mp2=1           PMT patch only 0x03/0x04->0x0F, zero CPU
    fix_interlace=1     H264 SPS progressive patch, zero CPU (LG 1080i fix)
    pts_reset=1         reset PTS/DTS to near-zero, zero CPU
    video_transcode=1   H264/MPEG2 -> H264 re-encode (~100% CPU) — only needed
                        when source is MPEG-2; NOT needed for H264 source
    video_crf=23        CRF quality for video_transcode (0-51, default 23)
    audio_bitrate=128000 AAC target bitrate for audio_transcode (default 128000)
    audio_map=N         Select ONLY the Nth audio track (1-indexed, by PMT
                        order) — every other audio track is dropped from
                        output entirely. Default (unset): select and
                        process ALL audio tracks found in the PMT — each
                        one independently transcoded to AAC if it's
                        MP2/MP3, passed through untouched if already AAC,
                        or passed through with a warning if it's a type
                        we can't decode (e.g. AC3).
    audio_coder=twoloop AAC encoder algorithm: "twoloop" (default, best
                        quality/bit) or "fast" (~2.4x lower CPU, slightly
                        less optimal bit allocation — measured via direct
                        benchmark, not a guess). Native ffmpeg AAC encoder
                        only — libfdk_aac is not available in this build,
                        and has no VBR mode (CBR/bit_rate only).

  EXAMPLES:
    # JAYA TV HD (H264 + MP2, interlaced 1080i):
    236.1.3.78 1026 /hls/jaya jaya 192.168.82.2 - - - 0  audio_transcode=1 fix_interlace=1

    # STAR MOVIES HD (H264 + MP2):
    237.1.1.17 1001 /hls/star star 192.168.81.2 - - - 0  audio_transcode=1

    # Old MPEG-2 SD channel:
    235.1.1.10 1000 /hls/old  old  192.168.81.2 - - - 0  audio_transcode=1 video_transcode=1 video_crf=23

  NOTE — this build is the original MP2Resync baseline with ONLY the
  automatic interface resolution feature grafted in (pass "-" for
  <iface> and the box figures out which NIC to join on per channel —
  see probe_iface_for_group()/resolve_iface_for_group()/auto_iface()
  and their use in sock_open(), plus the matching "-"-to-empty
  normalization added to the direct-CLI-argument parsing path in
  main()). sps_patch_interlace() (fix_interlace=1) is intentionally
  left as the original, unmodified baseline version — no SEI
  pic_struct neutralizer, no SPS bit-walk fixes — by request.
*/
#define _GNU_SOURCE
#include <stdio.h>
#include <stdlib.h>
#include <stdint.h>
#include <string.h>
#include <unistd.h>
#include <errno.h>
#include <pthread.h>
#include <signal.h>
#include <fcntl.h>
#include <time.h>
#include <poll.h>
#include <ifaddrs.h>
#include <net/if.h>
#include <sys/stat.h>
#include <sys/file.h>
#include <sys/socket.h>
#include <arpa/inet.h>
#include <netdb.h> /* getaddrinfo -- for the HLS-input HTTP client's
    hostname resolution (playlist/segment/key fetches)                  */
#include <netinet/in.h>
#include <openssl/aes.h>
#include <mpg123.h>
#include <libavcodec/avcodec.h>
#ifndef FF_PROFILE_H264_MAIN
#define FF_PROFILE_H264_MAIN AV_PROFILE_H264_MAIN
#endif
#include <libavutil/channel_layout.h>
#include <libavutil/frame.h>
#include <libavutil/opt.h>
#include <libavutil/imgutils.h>
#include <libavutil/log.h>
#include <libswresample/swresample.h>
#include <libswscale/swscale.h>
#include <math.h>

/* ════════════════════════════════════════════════════════════════
   VERSION
   ════════════════════════════════════════════════════════════════
   Single source of truth for "which build is this". To cut a new
   release, add a new VER_Vxx line to the enum (keep numeric order --
   the value IS the version number) and point VERSION_CURRENT at it.
   The startup banner, the no-args usage screen, and
   `udp_hls --version` all derive the printed tag from this enum, so
   it can never drift out of sync with what's actually running.      */

/* <<< Bump this to the newest AppVersion entry when cutting a release. */
#define VERSION_CURRENT "V47"

/* Short tag for the currently-built version, e.g. "V47". */
static const char *version_tag(void){
  	return "V47";
}

/* One-line identification string: version + exact build timestamp.
 * __DATE__/__TIME__ are filled in by the compiler at compile time, so
 * this always reflects exactly when THIS binary was built -- settles
 * "is this fix actually in the running binary" from a real log line
 * instead of guessing from a filename or from behavior alone.        */
static void print_version(void){
    printf("UDP->HLS IPTV Streaming Engine  %s  (build %s %s)\n",
           version_tag(),__DATE__,__TIME__);
}

/* ── tunables ──────────────────────────────────────────────────── */
#define SEGMENT_SECS   2
/* Hard ceiling on how long a single segment is allowed to stay open
 * while waiting for the next has_keyframe()/IDR cut point in simple-
 * copy/audio_transcode mode (see the segment-cut block in pkt_process).
 * Root cause this guards against, confirmed via real production log
 * analysis: this source's encoder occasionally goes far longer than
 * its normal cadence between functional-keyframe NALs (no CC drop
 * involved — cc_tot stayed flat across the whole event both times it
 * was observed), and without ANY timeout the segment just keeps
 * growing — confirmed real cases: two genuine 14.2-second segments
 * (7x the 2s target) with zero trace in cc_tot/idle/STALE, since none
 * of those track "expected keyframe is late", only "no packets at
 * all". A real player has no equivalent patience — this is exactly
 * the kind of gap that shows up as a visible freeze/black frame on
 * playback. Set well above SEGMENT_SECS so it only fires on a genuine
 * anomaly, never on normal jitter (the existing 10% tolerance already
 * handles that).                                                      */
#define MAX_SEG_WAIT_SECS (SEGMENT_SECS*4)
/* Hybrid IDR->PUSI fallback for the STARTUP (STATE 0) and CC-drop
 * RECOVERY (STATE 1) gates in simple-copy/audio_transcode mode — see
 * the fallback blocks in pkt_process for full rationale. Unlike the
 * MAX_SEG_WAIT_SECS cut timeout above (which only bounds an already-
 * PLAYING channel's segment length), these two gates had NO timeout
 * at all: can_start/can_recover require a genuine IDR (simple_cut=1)
 * or has_keyframe()'s Exp-Golomb functional-keyframe match (default),
 * and a source that never satisfies that condition on its very first
 * (or post-CC-drop) video PUSI packets got stuck there PERMANENTLY —
 * confirmed as the direct cause of channels showing NO_SIGNAL under
 * simple_cut=1 while the same channels play fine on v23, which starts
 * unconditionally on the first video PUSI with no IDR requirement at
 * all (v23's own IDR-awareness is scoped to segment CUT points only,
 * exactly like MAX_SEG_WAIT_SECS/idr_off above). Prefer a clean IDR/
 * keyframe start when the source provides one within this window;
 * only fall back to v23's unconditional-PUSI behavior once it's clear
 * the wait would otherwise be unbounded. Kept equal to MAX_SEG_WAIT_SECS
 * as a reasonable default (same order of magnitude as "how long is
 * too long to wait for a keyframe on this class of source"), but is
 * intentionally a separate constant from it since startup/recovery
 * and mid-stream cuts are different tradeoffs and may need to diverge
 * later.                                                              */
#define MAX_START_WAIT_SECS (SEGMENT_SECS*4)
/* v23-compatible permanent open-GOP fallback for the segment-CUT logic
 * (distinct from MAX_START_WAIT_SECS above, which only covers STARTUP/
 * RECOVERY). The existing MAX_SEG_WAIT_SECS forced-cut fallback fires
 * correctly on a genuine one-off keyframe-cadence anomaly, but on a
 * source that NEVER produces a cut point has_keyframe()/simple_cut can
 * detect (confirmed real case: every single segment forces the full
 * MAX_SEG_WAIT_SECS wait, back-to-back, indefinitely — e.g. 8.0s,
 * 8.1s, ... segments instead of the target 2s), paying that full
 * timeout on EVERY segment forever produces exactly the jerky, over-
 * long-segment behavior v23 does NOT have on the same source. v23
 * avoids this by permanently giving up on IDR-detection for a channel
 * (idr_off=1) after a small number of consecutive misses and cutting
 * on any subsequent video PUSI once normal segment timing allows —
 * see the is_cut_point/open_gop logic in the segment-cut block below.
 * Threshold is intentionally much smaller than v23's raw idr_misses=5
 * (which counts ~0.5s sub-waits within a single ~2s segment window,
 * i.e. engages after roughly 2.5-12.5s): here each "miss" already
 * costs a full MAX_SEG_WAIT_SECS-long segment, so reaching the same
 * number of misses would mean tens of seconds of oversized segments
 * before smoothing out.                                              */
#define OPEN_GOP_MISS_THRESHOLD 2
#define MAX_SEGMENTS   10
#define KEEP_EXTRA     3
#define MAX_CHANNELS   512
#define MAX_NICS       16
#define WRITE_BATCH    128
#define RCVBUF         (32<<20)
#define DUR_SLOTS      256
#define MAX_PER_CHAN   7
#define TS_SZ          188

/* ── PCR-aware gap stuffing — matches TSDuck's null-packet approach ──
 *
 * When a CC discontinuity is detected on the video/PCR PID in simple-
 * copy mode, we need to fill the gap with packets that maintain a
 * smooth, continuous PCR clock — not just null packets with PID=0x1FFF.
 *
 * Why PCR continuity matters: ExoPlayer (and all HLS-compliant players)
 * use the PCR as the master clock reference for A/V sync and buffer
 * management. If N packets are missing and we just write N PID=0x1FFF
 * null packets in their place, the PCR value in the NEXT REAL PACKET
 * (written by the encoder for wall-clock time T+N_packets) arrives
 * immediately after the nulls with NO time gap in the file — the
 * player's PCR clock jumps forward abruptly by the gap duration
 * (~1.91ms for a 7-packet gap at 5.5Mbps). This causes the decoder
 * to rush its output to catch up, producing visible drops/stutter.
 *
 * TSDuck solves this by writing the null packets with INTERPOLATED
 * PCR VALUES on the actual PCR PID (same PID as video, 0x0641 for
 * this stream) using adaptation-field-only packets (AFC=2). The PCR
 * advances smoothly across the gap so the decoder sees a continuous
 * clock and conceals the missing frames silently.
 *
 * Confirmed by direct comparison: same source through TSDuck→ffmpeg
 * plays clean in ExoPlayer, same source through udp_hls with plain
 * null stuffing still shows drops — the PCR discontinuity is the
 * remaining difference.
 *
 * ts_read_pcr(): extract the 27MHz PCR value from a TS packet that
 *   has an adaptation field with PCR_flag set (byte 5 bit 4 = 1).
 *   Returns -1 if this packet has no PCR.
 *
 * ts_write_pcr_pkt(): build a complete 188-byte adaptation-field-only
 *   TS packet on the given PID carrying the given 27MHz PCR value.
 *   Used to stuff interpolated PCR packets into the gap.             */

static int64_t ts_read_pcr(const uint8_t*p){
    /* TS header: p[0]=sync, p[1..2]=flags+PID, p[3]=CC+AFC
     * AFC field: bits 5-4 of p[3] = adaptation_field_control
     *   0x20 = AFC=2 (adaptation only)
     *   0x30 = AFC=3 (adaptation + payload)
     * adaptation_field starts at p[4]:
     *   p[4] = adaptation_field_length
     *   p[5] = flags byte: bit 4 = PCR_flag
     *   p[6..11] = PCR if PCR_flag set (48 bits total):
     *     base[32:0] in bits 47-15, marker=1 in bit 14 (actually
     *     the standard encoding is: base[32:25] in p[6],
     *     base[24:17] in p[7], base[16:9] in p[8],
     *     base[8:1] in p[9], base[0] in bit7 of p[10],
     *     reserved 6 bits (0x7E), ext[8] in bit0 of p[10],
     *     ext[7:0] in p[11].
     *     27MHz PCR = base*300 + ext                               */
    int afc=(p[3]>>4)&0x3;
    if(afc!=2&&afc!=3)return -1;        /* no adaptation field */
    if(p[4]<7)return -1;                /* adaptation field too short */
    if(!(p[5]&0x10))return -1;         /* PCR_flag not set */
    uint64_t base=(uint64_t)p[6]<<25|(uint64_t)p[7]<<17|
                  (uint64_t)p[8]<<9|(uint64_t)p[9]<<1|
                  ((p[10]>>7)&1);
    uint16_t ext=(uint16_t)((p[10]&1)<<8)|p[11];
    return (int64_t)(base*300+ext);}

static void ts_write_pcr_pkt(uint8_t*out,uint16_t pid,int64_t pcr27){
    /* Build a 188-byte adaptation-field-only TS packet (AFC=2) on
     * the given PID, carrying the given 27MHz PCR value.
     * Layout: 4-byte header + 184-byte adaptation field.
     * adaptation_field_length = 183 (rest of packet after the length
     * byte itself), flags byte with PCR_flag=1, 6 PCR bytes, then
     * stuffing bytes (0xFF) to fill the rest.                       */
    memset(out,0xFF,188);
    out[0]=0x47;
    out[1]=(uint8_t)(0x00|((pid>>8)&0x1F));
    out[2]=(uint8_t)(pid&0xFF);
    out[3]=0x20;                         /* AFC=2, CC=0 (no payload) */
    out[4]=183;                          /* adaptation_field_length  */
    out[5]=0x10;                         /* PCR_flag=1, rest=0       */
    uint64_t base=(uint64_t)pcr27/300;
    uint16_t ext=(uint16_t)(pcr27%300);
    out[6]=(uint8_t)((base>>25)&0xFF);
    out[7]=(uint8_t)((base>>17)&0xFF);
    out[8]=(uint8_t)((base>>9)&0xFF);
    out[9]=(uint8_t)((base>>1)&0xFF);
    out[10]=(uint8_t)(((base&1)<<7)|0x7E|((ext>>8)&1));
    out[11]=(uint8_t)(ext&0xFF);
    /* bytes 12..187 are already 0xFF (stuffing) from memset        */}

/* Null TS packet — PID=0x1FFF, payload-only, all-0xFF payload.
 * Used for non-PCR-PID gap filling (audio PID drops etc.) where
 * PCR restamping is not needed.                                      */
static const uint8_t NULL_TS_PKT[188]={
    0x47,0x1F,0xFF,0x10, /* sync, PID=0x1FFF, payload_only, CC=0 */
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF, /* 184 bytes of payload */
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF};
#define MAX_AUD_TRACKS 8       /* max simultaneously selected/transcoded
                                   audio tracks per channel - real
                                   broadcasts essentially never exceed
                                   a handful of audio tracks            */
#define AES_BLK        16
#define UDP_MAX        65536

/* ── utilities ─────────────────────────────────────────────────── */
static void ts_now(char *b){
    time_t t=time(NULL); struct tm *m=localtime(&t);
    snprintf(b,9,"%02d:%02d:%02d",m->tm_hour,m->tm_min,m->tm_sec);
}
static inline double wall_el(const struct timespec *a,const struct timespec *b){
    return (b->tv_sec-a->tv_sec)+(b->tv_nsec-a->tv_nsec)/1e9;
}

/* ── lock-free event ring ──────────────────────────────────────── */
#define EVT_RING 256
typedef enum{ EVT_NONE=0,EVT_PAT,EVT_PMT,EVT_START,
              EVT_OPEN,EVT_CLOSE,EVT_CC,EVT_STALE }EvtType;
typedef struct{ EvtType type;time_t when;uint64_t seq;
                uint32_t u32a,u32b;uint16_t u16a;uint8_t u8a,u8b;}Evt;
typedef struct{ Evt ring[EVT_RING];unsigned wr,rd;}EvtRing;
static inline void evpush(EvtRing*r,const Evt*e){
    unsigned w=__atomic_load_n(&r->wr,__ATOMIC_RELAXED);
    r->ring[w&(EVT_RING-1)]=*e;
    __atomic_store_n(&r->wr,w+1,__ATOMIC_RELEASE);}
static void evdrain(EvtRing*r,const char*n){
    unsigned w=__atomic_load_n(&r->wr,__ATOMIC_ACQUIRE);
    while(r->rd!=w){
        Evt*e=&r->ring[r->rd&(EVT_RING-1)];r->rd++;
        struct tm*m=localtime(&e->when);char t[9];
        snprintf(t,9,"%02d:%02d:%02d",m->tm_hour,m->tm_min,m->tm_sec);
        switch(e->type){
        case EVT_PAT:  printf("  [%s][%s] PAT pmt_pid=0x%04X\n",t,n,e->u16a);break;
        case EVT_PMT:  printf("  [%s][%s] PMT vid=0x%04X aud=0x%04X pcr=0x%04X\n",t,n,e->u16a,e->u32a,e->u32b);break;
        case EVT_START:printf("  [%s][%s] CLEAN START vid=0x%04X\n",t,n,e->u16a);break;
        case EVT_OPEN: printf("  [%s][%s] opened  index%llu.ts\n",t,n,(unsigned long long)e->seq);break;
        case EVT_CLOSE:printf("  [%s][%s] closed  index%llu.ts  %u.%03us  %ukbps\n",
                              t,n,(unsigned long long)e->seq,e->u32a/1000,e->u32a%1000,e->u32b);break;
        case EVT_CC:   printf("  [%s][%s] CC DROP pid=0x%04X exp=0x%X got=0x%X\n",t,n,e->u16a,e->u8a,e->u8b);break;
        case EVT_STALE:printf("  [%s][%s] STALE no input for %u.%03us\n",t,n,e->u32a/1000,e->u32a%1000);break;
        default:break;}}}

/* ════════════════════════════════════════════════════════════════
   AUDIO TRANSCODER  MP2/MP3 -> AAC-LC
   mpg123 decode -> PCM -> libavcodec AAC encode -> ADTS -> TS
   ════════════════════════════════════════════════════════════════ */
#define AAC_FRAME  1024
#define ADTS_HDR   7
static const int SR_TABLE[]={96000,88200,64000,48000,44100,32000,
                              24000,22050,16000,12000,11025,8000,7350};
typedef enum{AUDXC_SRC_MP2=0,AUDXC_SRC_AC3}AudXcSrc;
typedef struct{
    int             active;
    AudXcSrc        src_codec;    /* which decoder front-end is in use   */
    mpg123_handle  *mh;           /* MP2/MP3 front-end (src_codec==MP2)  */
    AVCodecContext *ac3_dec_ctx;  /* AC3/E-AC3 front-end (src_codec==AC3)*/
    AVPacket       *ac3_pkt;
    AVFrame        *ac3_frame;
    AVCodecParserContext *ac3_parser; /* finds AC3 frame boundaries across
        TS packets — a single AC3 frame typically spans several TS
        packets (confirmed: ~8 packets at 384kbps/48kHz), so feeding one
        packet's payload as "one complete frame" (the initial, broken
        attempt) fails decode with "Invalid data found" almost every
        time. The parser accumulates bytes across calls and hands back
        exactly one complete frame at a time, regardless of input
        chunking — the correct, standard way to handle this.          */
    struct SwrContext *swr;       /* fltp[+downmix] -> interleaved s16,
        allocated lazily once the AC3 stream's actual channel layout/
        sample rate is known from the first decoded frame             */
    int             swr_ready;
    AVCodecContext *ctx;
    AVFrame        *frame;
    AVPacket       *pkt;
    int             frame_size,sr_idx;
    int16_t         pcm[AAC_FRAME*2*4];
    int             pcm_fill;
    uint8_t         out[TS_SZ*32];/* 32 TS pkts: handles large AAC frames */
    /* adts_scratch/pcm_scratch/mp2_raw_scratch used to live HERE, one
     * copy per AudXc (i.e. per audio track, up to MAX_AUD_TRACKS=8 per
     * channel). Moved up to Ch (see Ch.aud_scratch) and shared across
     * all of a channel's tracks instead: audxc_push for a given
     * channel is always called sequentially from pkt_process's single-
     * threaded per-packet dispatch -- never two tracks' audxc_push
     * calls in flight at once for the SAME channel -- so one shared
     * set of scratch buffers per channel is exactly as safe as one per
     * track, at 1/8th the static memory cost (different channels run
     * on different NIC threads and each get their own Ch, hence their
     * own scratch buffers -- this is per-channel sharing, not a true
     * global). See AudScratch doc at its definition for full detail.  */
    int             out_len;
    int64_t         pts,pts_inc;
    int             pts_ok;
    int             pusi_seen;   /* discard continuation before first PUSI */
    uint64_t        push_calls;     /* total audxc_push calls with real
        data (srclen>0) since this slot was initialized - for the
        SIGUSR1 dump, to distinguish "genuinely stalled despite
        active=1" from other failure modes. Confirmed necessary: a
        real production dump showed active=1/pts_ok=1/pusi_seen=1
        (all healthy-looking) at the exact same moment 7 consecutive
        real segments showed zero audio packets.                      */
    uint64_t        push_calls_with_output; /* subset of push_calls
        where audxc_push actually returned nonzero output bytes.      */
    int64_t         last_seen_source_pts; /* most recent real PES PTS
        seen on this track (updated on every PUSI, not just the
        first) - see latency_comp doc for how this is used.          */
    int             last_seen_source_pts_valid;
    int64_t         samples_since_anchor; /* total PCM samples fed into
        the AAC encoder since pts_ok was set - used to compute the
        EXPECTED output pts at any point, for comparison against
        last_seen_source_pts (see latency_comp doc).                  */
    int             latency_measured;  /* 1 once latency_comp has been
        computed from a real measurement - measured once per channel
        start/CC-recovery, not every frame, since pipeline latency is
        structurally fixed for a given codec/decoder configuration.   */
    int64_t         latency_comp; /* measured pipeline latency, in
        90kHz PTS units, in the SOURCE's own timestamp domain — NOT
        wall-clock time. Computed as: (anchor_pts + samples_since_
        anchor_in_90khz) - last_seen_source_pts, sampled once a
        reasonable amount of source data has passed through. This
        directly answers "how far behind the source's own clock has
        our output PTS counter drifted, due to real decode+encode
        latency" — confirmed necessary via direct measurement: video
        runs ~60-100ms ahead of every transcoded audio track (AC3 and
        MP2 alike), present ONLY on audio_transcode channels (absent
        on passthrough-AAC channels), meaning this is real, physical
        pipeline latency that must be measured in the source's own
        timestamp domain, not wall-clock time (a wall-clock-based
        measurement was tried first and confirmed, via direct
        comparison against real captured output, to produce
        essentially no correction at all — CPU processing time and
        audio-timeline latency are different quantities).             */
    int64_t         anchor_pts; /* the source PTS value at the exact
        moment pts_ok was set — kept fixed and never modified again
        (unlike `pts` itself, which keeps advancing by pts_inc every
        frame and gets nudged by drift corrections below), so
        expected_src_pts can always be recomputed reliably from this
        fixed reference point plus samples_since_anchor, no matter how
        long the channel has been running or how many corrections have
        already been applied.                                         */
    int64_t         cumulative_drift_correction; /* running total of
        every correction (initial latency_comp plus every ongoing
        drift nudge since) already folded into `pts` so far — lets the
        ongoing correction below compute only the NEW, not-yet-applied
        drift at each check, instead of re-measuring and re-applying
        the same historical gap repeatedly.
        RESTORED: this and the ongoing per-PUSI correction below were
        removed in a merge; confirmed necessary again via direct A/B
        against a working ffmpeg audio-transcode reference on the same
        source over a genuinely long run (~2 hours) -- ffmpeg holds
        steady, this pipeline drifted to a consistent ~2.9s gap over
        that time. latency_comp above is a ONE-TIME correction and
        structurally cannot address this: real long-run drift comes
        from the small, continuous deviation between this source's
        true audio sample rate and the exact-48kHz rate pts_inc
        assumes (real oscillators are never exactly nominal), which
        only a correction that keeps running for the life of the
        channel can actually track down. Carries both regression fixes
        found the hard way when this was first built: each nudge is
        capped to pts_inc/16 (never large enough to push output PTS
        backward relative to the previously emitted frame, which
        caused a real full-audio-drop regression the first time this
        was tried without the cap), and the sanity bound on the
        measured gap is a wide +/-1 hour (only rejecting genuinely
        nonsensical values -- a real PTS wraparound or decode garbage
        -- not limiting how much real accumulated drift can be
        corrected; a tight bound here silently refused to correct
        exactly the channels that needed it most, the moment their gap
        exceeded it).                                                  */
}AudXc;

/* Scratch buffers shared by all of ONE channel's audio tracks (see the
 * comment in AudXc above for why this is safe: a channel's audxc_push
 * calls -- across however many of its tracks are active -- are always
 * sequential, never concurrent, since they're all driven from the
 * same single-threaded per-packet pkt_process dispatch). One of these
 * lives in Ch, not in AudXc/AudSlot, so the cost is paid once per
 * channel instead of once per track (up to MAX_AUD_TRACKS=8x more
 * before this change). Different channels run on different NIC
 * threads and each have their own Ch and thus their own AudScratch --
 * this is per-channel sharing, not a true cross-channel global.       */
typedef struct{
    uint8_t  adts[ADTS_HDR+1024]; /* see audxc_feed_pcm's use: sized for
        AAC-LC up to ~320kbps with headroom (1024 samples/frame @
        48kHz: even 320kbps CBR averages ~853 bytes/frame -- see
        audio_bitrate option doc for the supported range).            */
    int16_t  pcm[AAC_FRAME*5]; /* see audxc_push_ac3's use: sized for
        the actual worst-case AC3 resample output -- max AC3 frame is
        1536 samples/channel, worst-case upsampling from a 32kHz
        source to 48kHz gives 1536*1.5=2304 samples/channel, *2 for
        stereo interleaved = 4608 int16 values; this (5120) covers
        that with ~10% headroom for swr's internal delay buffer.      */
    uint8_t  mp2_raw[1152*2*2*2]; /* see audxc_push_mp2's use: sized for
        the real spec maximum -- MPEG-1 Layer II/III max samples/frame
        = 1152, stereo int16 = 2 bytes/sample -> 4608 bytes minimum;
        this (9216) gives 2x headroom.                                */
}AudScratch;

static void adts_hdr(uint8_t*h,int pay,int sri,int ch){
    int tot=pay+ADTS_HDR;
    h[0]=0xFF;h[1]=0xF1;
    h[2]=(uint8_t)(0x40|(sri<<2)|((ch>>2)&1));
    h[3]=(uint8_t)(((ch&3)<<6)|((tot>>11)&3));
    h[4]=(uint8_t)((tot>>3)&0xFF);
    h[5]=(uint8_t)(((tot&7)<<5)|0x1F);
    h[6]=0xFC;
}

/* Write ADTS frame into one or more TS packets.
 * First packet: PUSI=1 + PES header. Continuations: PUSI=0.
 * Returns total bytes written (multiple of TS_SZ).            */
static int adts_to_ts(uint8_t*dst,uint16_t pid,
                      const uint8_t*adts,int alen,int64_t pts,uint8_t*cc){
    int written=0,src_off=0,first=1;
    int pes_hdr=14; /* 9 fixed + 5 PTS */
    while(src_off<alen){
        uint8_t*pkt=dst+written;
        memset(pkt,0,TS_SZ); /* zero, not 0xFF — see padding note below */
        pkt[0]=0x47;
        pkt[1]=(uint8_t)((first?0x40:0x00)|((pid>>8)&0x1F));
        pkt[2]=(uint8_t)(pid&0xFF);
        int po;
        if(first){
            pkt[3]=(uint8_t)(0x10|(*cc&0x0F));
            (*cc)=(*cc+1)&0x0F;
            uint8_t*p=pkt+4;
            int plen=pes_hdr-6+alen;
            p[0]=0;p[1]=0;p[2]=1;p[3]=0xC0;
            p[4]=(uint8_t)(plen>>8);p[5]=(uint8_t)(plen&0xFF);
            p[6]=0x80;p[7]=0x80;p[8]=0x05;
            p[9] =(uint8_t)(0x21|((pts>>29)&0x0E));
            p[10]=(uint8_t)((pts>>22)&0xFF);
            p[11]=(uint8_t)(0x01|((pts>>14)&0xFE));
            p[12]=(uint8_t)((pts>>7)&0xFF);
            p[13]=(uint8_t)(0x01|((pts<<1)&0xFE));
            po=4+pes_hdr; first=0;
        } else {
            po=4;
            int remain=alen-src_off;
            if(remain<TS_SZ-4){
                /* Last packet of this ADTS frame and it won't fill the TS
                 * packet exactly. The OLD code memset() the whole packet
                 * to 0xFF and just copied the short payload on top,
                 * leaving a run of raw 0xFF bytes immediately after the
                 * real ADTS data — INSIDE the payload area of an AFC=01
                 * (payload-only) packet, where the TS spec defines every
                 * byte as stream data with no legal padding mechanism.
                 * A run of 0xFF 0xFF... decodes as a fake ADTS sync word
                 * (0xFF) + flags byte (0xFF) + length-field bytes that
                 * compute to 8191 (0x1FFF, the field's max value) — and
                 * ffmpeg's demuxer, resyncing after the genuine frame
                 * ends, locks onto this fake header and tries to read an
                 * 8191-byte "frame" that doesn't exist, corrupting every
                 * subsequent AAC frame in the segment. This exact pattern
                 * was confirmed byte-for-byte in real captured output.
                 * Fix: use a proper adaptation field with spec-legal
                 * stuffing_byte padding (AFC=11), identical to the fix
                 * already applied to the video path in vidxc_write().    */
                int pad=(TS_SZ-4)-remain;
                int afl=pad-1; if(afl<0)afl=0;
                pkt[3]=(uint8_t)(0x30|(*cc&0x0F)); /* AFC=11 */
                (*cc)=(*cc+1)&0x0F;
                pkt[4]=(uint8_t)afl;
                if(afl>0){
                    pkt[5]=0x00;
                    for(int i=1;i<afl;i++)pkt[5+i]=0xFF; /* legal stuffing_byte */
                }
                po=4+1+afl;
            } else {
                pkt[3]=(uint8_t)(0x10|(*cc&0x0F)); /* AFC=01: full payload */
                (*cc)=(*cc+1)&0x0F;
            }
        }
        int sp=TS_SZ-po,cp=alen-src_off;if(cp>sp)cp=sp;
        memcpy(pkt+po,adts+src_off,cp);
        src_off+=cp; written+=TS_SZ;
    }
    return written;
}

/* Buffer interleaved stereo int16 PCM samples (ns sample-pairs starting
 * at s) into AAC_FRAME-sized chunks, encode each full chunk via the
 * shared AAC encoder context, and packetize the result into ADTS+TS
 * appended to a->out[]. Shared by both decoder front-ends (MP2 via
 * mpg123, AC3/E-AC3 via libavcodec) since the encode side is identical
 * regardless of source codec — only how PCM samples are produced
 * differs.                                                            */
static void audxc_feed_pcm(AudXc*a,AudScratch*scr,const int16_t*s,int ns,
                           uint16_t pid,uint8_t*cc){
    int so=0;
    while(so<ns){
        int sp=AAC_FRAME-a->pcm_fill,cp=ns-so;if(cp>sp)cp=sp;
        for(int i=0;i<cp;i++){
            a->pcm[(a->pcm_fill+i)*2+0]=s[(so+i)*2+0];
            a->pcm[(a->pcm_fill+i)*2+1]=s[(so+i)*2+1];
        }
        a->pcm_fill+=cp;so+=cp;
        if(a->pcm_fill>=AAC_FRAME){
            av_frame_make_writable(a->frame);
            float*L=(float*)a->frame->data[0];
            float*R=(float*)a->frame->data[1];
            for(int i=0;i<AAC_FRAME;i++){
                L[i]=a->pcm[i*2+0]/32768.0f;
                R[i]=a->pcm[i*2+1]/32768.0f;
            }
            a->pcm_fill=0;a->samples_since_anchor+=AAC_FRAME;
            if(!a->pts_ok){a->pts=0;a->pts_ok=1;}
            if(!a->latency_measured&&a->last_seen_source_pts_valid){
                /* Measure real pipeline latency entirely in the
                 * source's own PTS domain — never wall-clock time (a
                 * wall-clock measurement was tried first and confirmed,
                 * via direct comparison against real captured output,
                 * to produce essentially no correction at all on a
                 * normally-loaded system, since CPU processing time
                 * and audio-timeline latency are different
                 * quantities). a->pts is the real absolute source PTS
                 * captured at anchor time; samples_since_anchor is how
                 * much output-time we've produced since then. If the
                 * pipeline had zero latency, the source's clock would
                 * now read exactly a->pts + that many 90kHz units. The
                 * ACTUAL source clock (last_seen_source_pts, updated on
                 * every PUSI since) has advanced further than that —
                 * the difference is the real, physical decode+encode
                 * latency this output frame is behind by. Confirmed
                 * necessary via direct measurement: video runs a
                 * consistent ~60-100ms ahead of every transcoded audio
                 * track (AC3 and MP2 alike), present ONLY on
                 * audio_transcode channels.                            */
                int64_t expected_src_pts=a->pts+
                    (a->samples_since_anchor*90000LL/48000);
                int64_t gap=a->last_seen_source_pts-expected_src_pts;
                /* Sanity bound was originally 90000 (1.0s), on the
                 * assumption that real pipeline latency "should never be
                 * anywhere near a full second" — confirmed WRONG by
                 * direct measurement on a real audio_transcode capture:
                 * actual real latency was ~1.85-1.95s (171000-175000 in
                 * 90kHz units), consistently, across 12+ seconds of
                 * output — i.e. genuinely close to double the old cap,
                 * not a bogus/wrapped PTS. With the old 90000 bound,
                 * gap>=90000 was silently rejected outright (a->pts left
                 * uncorrected), so EVERY AAC frame was stamped ~1.9s
                 * earlier than it should have been — exactly the
                 * consistent, non-drifting video-ahead-of-audio offset
                 * observed in that capture. Raised to a still-bounded
                 * but much more realistic 5s (450000): large enough for
                 * genuinely slow decode/encode pipelines under real
                 * load, still far short of the ~26.5h 33-bit PTS wrap
                 * period, so it can't mistake a wrap for latency.        */
                if(gap>0&&gap<450000){
                    a->latency_comp=gap;
                    a->pts+=a->latency_comp; /* shift the base FORWARD:
                        this output is genuinely `gap` 90kHz-units
                        behind where the source's real clock already
                        is, so its correct presentation time is later
                        than the raw anchor would suggest — matching
                        how much decode+encode latency actually
                        elapsed before this frame became available.   */
                    a->cumulative_drift_correction+=a->latency_comp;
                    /* folds this one-time correction into the same
                     * running total the ongoing drift check (in
                     * pkt_process, on every audio PUSI) uses as its
                     * "already applied" baseline — without this, the
                     * very first ongoing check after startup would
                     * see this whole latency_comp as if it were fresh,
                     * unexplained drift and try to "correct" it again. */
                    if(gap>=90000)
                        fprintf(stderr,"WARNING: audio_transcode pipeline "
                                "latency measured at %.3fs (compensated) — "
                                "unusually high; if this keeps happening, "
                                "investigate decode/encode buffering or "
                                "system load rather than relying on this "
                                "one-time correction alone\n",gap/90000.0);
                }
                a->latency_measured=1;
            }
            /* NOTE: ongoing (not just one-time) audio clock drift
             * correction is applied elsewhere, in pkt_process, on
             * every audio PUSI — see the comment at
             * "s->audxc.last_seen_source_pts=src_pts" below in the
             * audio_transcode PUSI handler. Doing it there (driven by
             * real PES PTS arrivals) rather than here (driven by AAC
             * frame cadence) means the correction tracks the actual
             * source clock updates directly, with no extra bookkeeping
             * needed in this function.                                 */
            a->frame->pts=a->pts; a->pts+=a->pts_inc;
            if(avcodec_send_frame(a->ctx,a->frame)<0)continue;
            while(avcodec_receive_packet(a->ctx,a->pkt)==0){
                uint8_t*adts=scr->adts;
                int pay=a->pkt->size;
                if(pay>(int)sizeof(scr->adts)-ADTS_HDR)pay=(int)sizeof(scr->adts)-ADTS_HDR;
                adts_hdr(adts,pay,a->sr_idx,2);
                memcpy(adts+ADTS_HDR,a->pkt->data,pay);
                int alen=ADTS_HDR+pay;
                if(a->out_len+TS_SZ*3<=(int)sizeof(a->out)){
                    int wrote=adts_to_ts(a->out+a->out_len,pid,
                                         adts,alen,a->frame->pts,cc);
                    a->out_len+=wrote;
                }
                av_packet_unref(a->pkt);
            }
        }
    }
}

static int audxc_init(AudXc*a,int bitrate,const char*coder,AudXcSrc src_codec){
    memset(a,0,sizeof*a);
    a->src_codec=src_codec;
    if(src_codec==AUDXC_SRC_MP2){
        mpg123_init();
        int err; a->mh=mpg123_new(NULL,&err);
        if(!a->mh)return 0;
        mpg123_param(a->mh,MPG123_FLAGS,MPG123_QUIET,0);
        mpg123_format_none(a->mh);
        mpg123_format(a->mh,48000,MPG123_STEREO,MPG123_ENC_SIGNED_16);
        if(mpg123_open_feed(a->mh)!=MPG123_OK){mpg123_delete(a->mh);return 0;}
    } else { /* AUDXC_SRC_AC3 */
        const AVCodec*dec=avcodec_find_decoder(AV_CODEC_ID_AC3);
        if(!dec)dec=avcodec_find_decoder(AV_CODEC_ID_EAC3);
        if(!dec)return 0;
        a->ac3_dec_ctx=avcodec_alloc_context3(dec);
        if(avcodec_open2(a->ac3_dec_ctx,dec,NULL)<0){
            avcodec_free_context(&a->ac3_dec_ctx);return 0;}
        a->ac3_pkt=av_packet_alloc();
        a->ac3_frame=av_frame_alloc();
        a->ac3_parser=av_parser_init((int)dec->id);
        if(!a->ac3_parser){
            avcodec_free_context(&a->ac3_dec_ctx);
            av_packet_free(&a->ac3_pkt);av_frame_free(&a->ac3_frame);
            return 0;}
    }
    const AVCodec*codec=avcodec_find_encoder(AV_CODEC_ID_AAC);
    if(!codec){
        if(a->mh)mpg123_delete(a->mh);
        if(a->ac3_dec_ctx)avcodec_free_context(&a->ac3_dec_ctx);
        if(a->ac3_pkt)av_packet_free(&a->ac3_pkt);
        if(a->ac3_frame)av_frame_free(&a->ac3_frame);
        return 0;}
    a->ctx=avcodec_alloc_context3(codec);
    a->ctx->sample_rate=48000; a->ctx->bit_rate=bitrate>0?bitrate:128000;
    av_channel_layout_default(&a->ctx->ch_layout,2); /* AAC output is
        always stereo regardless of source — AC3 5.1 sources are
        downmixed to stereo when PCM is extracted in audxc_push.      */
    a->ctx->sample_fmt=AV_SAMPLE_FMT_FLTP; /* all AAC encoders use FLTP */
    if(coder&&coder[0])av_opt_set(a->ctx->priv_data,"aac_coder",coder,0);
    if(avcodec_open2(a->ctx,codec,NULL)<0){
        avcodec_free_context(&a->ctx);
        if(a->mh)mpg123_delete(a->mh);
        if(a->ac3_dec_ctx)avcodec_free_context(&a->ac3_dec_ctx);
        if(a->ac3_pkt)av_packet_free(&a->ac3_pkt);
        if(a->ac3_frame)av_frame_free(&a->ac3_frame);
        return 0;}
    a->frame_size=a->ctx->frame_size;
    a->frame=av_frame_alloc();
    a->frame->nb_samples=a->frame_size;
    a->frame->format=a->ctx->sample_fmt;
    av_channel_layout_copy(&a->frame->ch_layout,&a->ctx->ch_layout);
    av_frame_get_buffer(a->frame,0);
    a->pkt=av_packet_alloc();
    a->sr_idx=3; /* 48kHz */
    for(int i=0;i<(int)(sizeof SR_TABLE/sizeof*SR_TABLE);i++)
        if(SR_TABLE[i]==48000){a->sr_idx=i;break;}
    a->pts_inc=(int64_t)a->frame_size*90000/48000; /* =1920 */
    a->active=1;
    if(src_codec==AUDXC_SRC_AC3)
        printf("[audxc] AC3->AAC 48kHz stereo (downmixed if needed) %dkbps ready\n",
               bitrate>0?bitrate/1000:128);
    else
        printf("[audxc] MP2->AAC 48kHz stereo 128kbps ready\n");
    return 1;
}
static void audxc_free(AudXc*a){
    if(!a->active)return;
    av_frame_free(&a->frame);av_packet_free(&a->pkt);
    avcodec_free_context(&a->ctx);
    if(a->src_codec==AUDXC_SRC_MP2){
        mpg123_delete(a->mh);mpg123_exit();
    } else {
        av_frame_free(&a->ac3_frame);av_packet_free(&a->ac3_pkt);
        avcodec_free_context(&a->ac3_dec_ctx);
        if(a->ac3_parser)av_parser_close(a->ac3_parser);
        if(a->swr)swr_free(&a->swr);
    }
    a->active=0;
}
/* Feed raw MP2 payload bytes, produce AAC TS packets in a->out[].
 * Returns bytes written (multiple of TS_SZ).                  */
/* MP2/MP3 source: existing mpg123-based decode, now feeding the shared
 * audxc_feed_pcm helper instead of inline duplicate buffering/encode
 * code (no behavior change from before — pure refactor).              */
static int audxc_push_mp2(AudXc*a,AudScratch*scr,const uint8_t*mp2,int mp2len,
                          uint16_t pid,uint8_t*cc){
    mpg123_feed(a->mh,mp2,(size_t)mp2len);
    uint8_t*raw=scr->mp2_raw;
    size_t done; int ret;
    while((ret=mpg123_read(a->mh,raw,sizeof scr->mp2_raw,&done))==MPG123_OK
          ||ret==MPG123_NEW_FORMAT||ret==MPG123_NEED_MORE){
        if(!done){
            /* MPG123_NEW_FORMAT fires with done=0 the instant mpg123 locks
             * sync — the PCM for the frame that triggered the lock is
             * fetched on the *next* read call, not this one. Don't break;
             * loop again immediately to drain it in the same push.        */
            if(ret==MPG123_NEW_FORMAT) continue;
            /* MPG123_NEED_MORE with done=0: truly no more PCM buffered,
             * decoder is waiting for the next mpg123_feed(). Stop here. */
            break;
        }
        /* CRITICAL: MPG123_NEED_MORE can carry done>0 — mpg123 hands back
         * PCM for a fully-decoded frame in the SAME call that also says
         * "I need more input for the *next* frame". Discarding this case
         * (old code's loop condition excluded NEED_MORE entirely) silently
         * dropped every other frame's audio — net result was zero AAC ever
         * produced. We must process this data exactly like MPG123_OK,
         * then stop (no more buffered data left after a NEED_MORE).      */
        int16_t*s=(int16_t*)raw;
        int ns=(int)(done/sizeof(int16_t))/2;
        audxc_feed_pcm(a,scr,s,ns,pid,cc);
    }
    /* MPG123_ERR (or any return code outside the loop's accepted set)
     * means the decoder's internal stream state is desynced — typically
     * triggered by real upstream packet loss (a CC drop) corrupting the
     * MPEG audio frame boundary mpg123 was tracking. Confirmed as the
     * root cause of a real production incident via direct SIGUSR1
     * dump comparison: push_calls climbed by 14252 over 96s/48 segments
     * while push_calls_with_output, write_loop_entries and audxc.pts
     * stayed perfectly flat the entire time — mpg123_feed() never
     * errors (it only appends to its internal buffer), so packets kept
     * arriving and being fed in, but mpg123_read() was silently
     * returning MPG123_ERR on every single call forever after, with no
     * path back to a working decode state. The PMT kept advertising an
     * audio track throughout, and a production segment from this same
     * incident showed the audio PID completely absent (0 packets out of
     * ~6900 TS packets, verified by direct binary parse), matching the
     * ~96s of total silence. Closing and reopening the feed
     * (mpg123_close/mpg123_open_feed) resets mpg123's internal sync
     * state without tearing down the whole AudXc (no need to reallocate
     * the AAC encoder context, which is independent of the MP2 decode
     * front-end) — exactly the same "force a resync after real packet
     * loss" approach already used on the video CC-drop path
     * (force_idr+au_reset), just applied to the MP2 decoder instead of
     * the H264 decoder. We deliberately do NOT retry mpg123_read()
     * again in this same call: the loop above has already drained
     * everything decodable from this feed; any bytes still sitting in
     * the now-closed handle's internal buffer are discarded along with
     * it, which is correct — they belong to the corrupted stream state
     * we are resyncing away from, identical to how a video CC drop
     * discards the in-flight AU rather than trying to salvage it.      */
    if(ret==MPG123_ERR){
        static time_t _last_warn=0; time_t now_t=time(NULL);
        if(now_t!=_last_warn){
            fprintf(stderr,"[audxc] MPG123_ERR — MP2 decoder desynced "
                    "(likely real upstream packet loss), resetting feed\n");
            _last_warn=now_t;}
        mpg123_close(a->mh);
        if(mpg123_open_feed(a->mh)!=MPG123_OK){
            /* Reopen itself failing is a deeper problem (e.g. mpg123
             * internal allocation failure) — mark this track inactive
             * rather than spin forever calling a handle that can't even
             * accept feed data.                                         */
            fprintf(stderr,"[audxc] mpg123_open_feed FAILED on reset — "
                    "disabling this audio track\n");
            a->active=0;
        }
        /* pts_ok/pusi_seen/latency_measured/samples_since_anchor reset
         * exactly like a CC-drop recovery (see pkt_process's audio CC
         * handling) — the anchor PTS this track had is now meaningless
         * since the encoder will resume from a fresh PCM stream with no
         * continuity to what came before.                                */
        a->pts_ok=0; a->pusi_seen=0;
        a->last_seen_source_pts_valid=0;
        a->latency_measured=0; a->samples_since_anchor=0;
        a->anchor_pts=0; a->cumulative_drift_correction=0;
    }
    return a->out_len;
}

/* AC3/E-AC3 source: decode via libavcodec, downmix to stereo (if the
 * source is 5.1/surround) and convert fltp->s16 via swresample, then
 * feed the shared PCM pipeline. Unlike MP2/mpg123 (a streaming decoder
 * fed arbitrary-sized chunks), AC3 is frame-oriented — each TS PES
 * payload we're handed here is expected to be one complete AC3 sync
 * frame (this matches how the caller in pkt_process delivers it: a
 * full audio PES payload per call, same as the MP2 path receives).    */
static int audxc_push_ac3(AudXc*a,AudScratch*scr,const uint8_t*ac3,int ac3len,
                          uint16_t pid,uint8_t*cc){
    const uint8_t*in_data=ac3;int in_len=ac3len;
    while(in_len>0){
        uint8_t*frame_data=NULL;int frame_size=0;
        int consumed=av_parser_parse2(a->ac3_parser,a->ac3_dec_ctx,
            &frame_data,&frame_size,in_data,in_len,
            AV_NOPTS_VALUE,AV_NOPTS_VALUE,0);
        if(consumed<0)break; /* genuine parser error — stop, avoid
                                 an infinite loop on malformed input.
                                 consumed==0 is NOT an error: it means
                                 the parser handed back a complete frame
                                 it had already buffered, without
                                 needing to consume any new bytes this
                                 call — that frame must still be
                                 processed below, not discarded. (This
                                 distinction was the actual bug found
                                 via direct trace: real, valid AC3
                                 frames — confirmed 1536 bytes, matching
                                 384kbps/48kHz exactly — were being
                                 silently discarded every time because
                                 consumed==0 was wrongly treated as a
                                 stuck/error state and broke the loop
                                 before the frame was ever decoded.)    */
        in_data+=consumed;in_len-=consumed;
        if(frame_size<=0){
            if(consumed==0)break; /* truly no progress AND no frame —
                                      now it's safe to stop            */
            continue; /* parser consumed bytes but hasn't accumulated
                          a full frame yet — normal, keep feeding      */
        }
        av_packet_unref(a->ac3_pkt);
        /* Point directly at the parser's own buffer instead of
         * allocating a fresh one and copying into it — frame_data is
         * valid for the duration of this call, and avcodec_send_packet
         * only reads from the packet synchronously without retaining
         * it afterward, so no copy is actually needed here. This
         * removes one malloc+memcpy+free cycle per AC3 frame (~31x/sec
         * per active AC3/E-AC3 track at common bitrates) — a real,
         * measurable cost when multiple audio tracks run alongside
         * video decode/encode on this codebase's single-threaded
         * per-channel design.                                         */
        a->ac3_pkt->data=frame_data;a->ac3_pkt->size=frame_size;
        if(avcodec_send_packet(a->ac3_dec_ctx,a->ac3_pkt)<0)continue;
        while(avcodec_receive_frame(a->ac3_dec_ctx,a->ac3_frame)==0){
            if(!a->swr_ready){
                /* Lazily build the resampler now that we know the AC3
                 * stream's actual channel layout and sample rate — AC3
                 * streams can be mono, stereo, 5.1, etc, and this isn't
                 * known until the first frame is decoded.              */
                AVChannelLayout out_layout;
                av_channel_layout_default(&out_layout,2); /* stereo out */
                int rc=swr_alloc_set_opts2(&a->swr,
                    &out_layout,AV_SAMPLE_FMT_S16,48000,
                    &a->ac3_frame->ch_layout,(enum AVSampleFormat)a->ac3_frame->format,
                    a->ac3_frame->sample_rate,0,NULL);
                if(rc>=0&&a->swr)rc=swr_init(a->swr);
                av_channel_layout_uninit(&out_layout);
                if(rc<0||!a->swr){av_frame_unref(a->ac3_frame);continue;}
                a->swr_ready=1;
                printf("[audxc] AC3 source: %d ch, %dHz -> downmixing to stereo 48kHz\n",
                       a->ac3_frame->ch_layout.nb_channels,a->ac3_frame->sample_rate);
            }
            /* swr_convert may need to resample (if source isn't 48kHz) —
             * size the output buffer generously for that case.         */
            int max_out_samples=av_rescale_rnd(
                swr_get_delay(a->swr,a->ac3_frame->sample_rate)+a->ac3_frame->nb_samples,
                48000,a->ac3_frame->sample_rate,AV_ROUND_UP);
            int16_t*pcmbuf=scr->pcm; /* persistent, shared per-channel scratch */
            if(max_out_samples*2>(int)(sizeof(scr->pcm)/sizeof(int16_t)))
                max_out_samples=(int)(sizeof(scr->pcm)/sizeof(int16_t))/2;
            uint8_t*out_planes[1]={(uint8_t*)pcmbuf};
            int got=swr_convert(a->swr,out_planes,max_out_samples,
                                (const uint8_t**)a->ac3_frame->extended_data,
                                a->ac3_frame->nb_samples);
            av_frame_unref(a->ac3_frame);
            if(got>0)audxc_feed_pcm(a,scr,pcmbuf,got,pid,cc);
        }
    }
    return a->out_len;
}

static int audxc_push(AudXc*a,AudScratch*scr,const uint8_t*src,int srclen,
                      uint16_t pid,uint8_t*cc){
    a->out_len=0;
    if(!a->active||srclen<=0)return 0;
    a->push_calls++;
    int result=(a->src_codec==AUDXC_SRC_MP2)?
        audxc_push_mp2(a,scr,src,srclen,pid,cc):
        audxc_push_ac3(a,scr,src,srclen,pid,cc);
    if(result>0)a->push_calls_with_output++;
    return result;
}

/* ════════════════════════════════════════════════════════════════
   VIDEO TRANSCODER (optional, only for MPEG-2 source channels)
   H264/MPEG-2 -> libavcodec decode -> libavcodec/libx264 encode
   ════════════════════════════════════════════════════════════════ */
#define AU_MAX     (1<<20)
#define VOUT_MAX   8192

static enum AVCodecID st2codec(uint8_t st){
    switch(st){case 0x01:case 0x02:return AV_CODEC_ID_MPEG2VIDEO;
    case 0x1B:return AV_CODEC_ID_H264;case 0x24:return AV_CODEC_ID_HEVC;
    default:return AV_CODEC_ID_NONE;}}

typedef struct{uint8_t buf[AU_MAX];int len;int64_t pts,dts;int pts_valid;
    int truncated; /* set by au_push if this AU would have exceeded
        AU_MAX -- previously the overflow was silently truncated and
        the partial AU still got fed to the decoder, which can corrupt
        the decode (a real H264 AU cut off mid-slice-data, not at a
        NAL boundary, can desync the entropy decoder, corrupt SPS/PPS
        state for the picture, or produce random macroblocks). Now:
        the AU is discarded entirely (see vidxc_push's check of this
        flag right before sending to the decoder) and dec_ready is
        reset so the next genuine I-slice re-validates cleanly, same
        recovery path already used for a CC-drop or decode error --
        never partially decode a truncated AU.                        */
}AuBuf;
static void au_reset(AuBuf*a){a->len=0;a->pts_valid=0;a->truncated=0;}

/* Decode the Exp-Golomb slice_type field from a raw (non-emulation-
 * stripped) H264 slice NAL payload, to determine if this is a true
 * I-slice (safe random-access point) rather than just checking for
 * SPS presence elsewhere in the AU. Some real broadcast encoders
 * repeat SPS/PPS on a schedule independent of GOP/IDR boundaries, so
 * "AU contains an SPS NAL" does NOT reliably mean "this AU's picture
 * is safe to decode after a gap" — it can just as easily be an
 * ordinary P-slice that happens to be adjacent to a repeated SPS.
 * `payload` should point just after the NAL header byte; `len` is
 * the number of bytes available (emulation-prevention bytes are
 * handled inline). Returns 1 if this is an I-slice (slice_type 2 or
 * 7), 0 otherwise (including on any parse failure — fail safe).    */
static int is_islice(const uint8_t*payload,int len){
    if(len<=0)return 0;
    /* Strip emulation prevention (00 00 03 -> 00 00) into a small
     * local buffer; slice headers are short, a few dozen bytes is
     * always enough to reach first_mb_in_slice + slice_type.       */
    uint8_t buf[64];int blen=0;
    for(int i=0;i<len&&blen<(int)sizeof(buf);i++){
        if(i+2<len&&payload[i]==0&&payload[i+1]==0&&payload[i+2]==3){
            buf[blen++]=0;if(blen<(int)sizeof(buf))buf[blen++]=0;i+=2;
        }else buf[blen++]=payload[i];
    }
    if(blen<2)return 0;
    uint64_t bitpos=0;
    uint64_t totalbits=(uint64_t)blen*8;
    int err=0;
    int(*read_bit)(uint8_t*,uint64_t*,uint64_t,int*)=NULL;(void)read_bit;
    #define BR_BIT() ({ \
        int _v=0; \
        if(bitpos>=totalbits){err=1;} \
        else{ _v=(buf[bitpos/8]>>(7-(bitpos%8)))&1; bitpos++; } \
        _v; })
    #define BR_UE() ({ \
        int _lz=0; \
        while(!err&&BR_BIT()==0){_lz++;if(_lz>32){err=1;break;}} \
        int _r=0; \
        if(!err){ if(_lz>0){ for(int _k=0;_k<_lz&&!err;_k++)_r=(_r<<1)|BR_BIT(); _r=(1<<_lz)-1+_r; } } \
        _r; })
    (void)BR_UE(); /* first_mb_in_slice — discarded */
    int slice_type=BR_UE();
    #undef BR_BIT
    #undef BR_UE
    if(err)return 0;
    int st=slice_type%5; /* slice_type 5-9 mean "all slices in picture
                             are this type"; mod 5 normalizes 0-4      */
    return st==2; /* 2 == I-slice */
}

/* MPEG2 equivalent of is_islice() above -- completely different
 * bitstream grammar, so this is a separate function, not a variant of
 * the H.264 one. Per ISO/IEC 13818-2: a picture_start_code is the
 * 4-byte sequence 00 00 01 00 (the 4th byte, 0x00, is what
 * distinguishes it from slice_start_codes 00 00 01 01..AF, and from
 * other start codes like sequence_header 00 00 01 B3). Immediately
 * following the start code, picture_header begins with:
 *   temporal_reference   : 10 bits
 *   picture_coding_type  :  3 bits  (1=I, 2=P, 3=B, 4=D)
 *   vbv_delay            : 16 bits
 * ...packed into the first 2 bytes after the start code as:
 *   byte0[7:0] ++ byte1[7:0] = TTTTTTTT TTCCCVVV
 *   (T=temporal_reference, C=picture_coding_type, V=vbv_delay high bits)
 * No Exp-Golomb, no emulation-prevention bytes in this region (that's
 * an H.264/HEVC-only concept) -- this is a fixed-width bit-field read,
 * much simpler than the H.264 case.                                   */
static int is_mpeg2_iframe(const uint8_t*d,int sz){
    if(sz<6)return 0;
    for(int i=0;i+5<sz;i++){
        if(d[i]==0&&d[i+1]==0&&d[i+2]==1&&d[i+3]==0x00){
            /* picture_start_code found -- decode picture_coding_type */
            uint16_t hdr16=((uint16_t)d[i+4]<<8)|d[i+5];
            int picture_coding_type=(hdr16>>3)&0x7;
            return picture_coding_type==1; /* 1 == I-frame */
        }
    }
    return 0;
}

/* MPEG2 decode has a prerequisite the H.264 gate above doesn't need to
 * think about separately: a sequence_header carries frame dimensions,
 * chroma format, and other sequence-level parameters the decoder must
 * have before it can decode ANY picture at all -- not just before
 * I-frames specifically. sequence_header_code is 00 00 01 B3.
 * FIX: the original I-slice gate only forwarded AUs containing a real
 * I-frame (picture_coding_type==1) to the decoder, discarding every
 * other AU while dec_ready==0 -- including any AU that carried ONLY
 * the sequence_header with no picture data. If this source sends the
 * sequence_header as its own separate access unit (not bundled with
 * the I-frame's picture data in the same AU), it was being silently
 * discarded before ever reaching the decoder -- so even once a real
 * I-frame later arrived and got forwarded, the decoder still lacked
 * the sequence-level context it needs, and rejected it as invalid
 * data. This directly matches two independent pieces of evidence: the
 * very first report in this investigation (ffmpeg's own decoder log:
 * "Invalid frame dimensions 0x0" -- the same missing-sequence-header
 * symptom, different decoder), and this codebase's own repeated
 * send_packet AVERROR_INVALIDDATA failures immediately after "I-slice
 * found" on a real MPEG2 channel. Forward sequence_header-bearing AUs
 * to the decoder too, not just I-frame-bearing ones.                  */
static int has_mpeg2_seq_header(const uint8_t*d,int sz){
    for(int i=0;i+3<sz;i++){
        if(d[i]==0&&d[i+1]==0&&d[i+2]==1&&d[i+3]==0xB3)return 1;
    }
    return 0;
}

/* True if this elementary-stream payload (the bytes of a PES packet
 * immediately after its header has been stripped) begins with a start
 * code that legitimately marks the beginning of a brand new MPEG2
 * access unit: picture_start_code (0x00), sequence_header_code
 * (0xB3), or group_start_code (0xB8) -- a sequence/GOP header always
 * immediately precedes the picture it belongs to, so treating either
 * as an AU boundary too is correct, not just picture_start_code on
 * its own.
 *
 * Anything else here -- a slice_start_code (0x01-0xAF), an
 * extension_start_code (0xB5) continuing the previous picture's
 * extension data, or plain slice bytes with no start code at all --
 * is a CONTINUATION of the picture already in progress, even though
 * it arrived in its own, separately PUSI-flagged PES packet.
 *
 * Confirmed via direct dump analysis on a real production channel:
 * this source's muxer splits a single access unit's elementary-
 * stream data across multiple PES packets (~2000-byte chunks), each
 * with its own PUSI. Treating every PUSI as an AU boundary (the old
 * behavior -- unconditional a->len=0 in au_push) silently truncated
 * every picture to whatever fit in the FIRST chunk and fed the
 * leftover chunks to the decoder as bogus, headerless "next AUs" --
 * 15 of 16 captured chunks in that dump started mid-slice (first
 * start code found was a slice_start_code, not a picture header) and
 * every failing AU's total size (2002/2002/1997 bytes) matched one
 * chunk, never a real, complete access unit (confirmed >=4497 bytes
 * via raw capture).                                                    */
static int mpeg2_is_au_start(const uint8_t*p,int n){
    if(n<4||p[0]||p[1]||p[2]!=1)return 0;
    return p[3]==0x00||p[3]==0xB3||p[3]==0xB8;
}

/* Strip a PUSI-flagged TS packet's adaptation field and full PES
 * header, returning a pointer to the elementary-stream payload right
 * after it plus its length. Returns 0 if this isn't a parseable
 * PES-start packet. Shared by vidxc_push (to decide whether a given
 * PUSI is a real AU boundary or a continuation, BEFORE deciding
 * whether to flush the AU accumulated so far) and au_push (which
 * re-derives the same payload to actually copy/extract pts+dts from
 * it) -- both must agree on where a new AU truly starts, or the
 * flush-gate and the accumulate-gate can disagree with each other
 * exactly like the previous, incomplete version of this fix did.      */
static int ts_pes_es_payload(const uint8_t*pkt,const uint8_t**out_pay,int*out_plen){
    int afc=(pkt[3]>>4)&3;if(afc==2)return 0;
    int off=4;if(afc==3)off+=1+(int)pkt[4];if(off>=TS_SZ)return 0;
    const uint8_t*pay=pkt+off;int plen=TS_SZ-off;
    if(plen<9||pay[0]!=0||pay[1]!=0||pay[2]!=1)return 0;
    int skip=9+pay[8];pay+=skip;plen-=skip;
    if(skip>TS_SZ)return 0;
    *out_pay=pay;*out_plen=plen;
    return 1;
}

/* Push one TS packet into AU accumulator.
 * CRITICAL: discard continuation packets before first PUSI
 * (prevents AVERROR_INVALIDDATA from mid-AU decoder feed).
 * is_au_boundary: precomputed by the caller (vidxc_push), using the
 * SAME ts_pes_es_payload()+mpeg2_is_au_start() check that decides
 * whether to flush the in-progress AU. Only meaningful when this
 * packet is PUSI-flagged (au_push re-derives pusi from pkt itself for
 * the PES-header-strip either way); ignored on continuation (non-
 * PUSI) TS packets, which always append to whatever AU is already in
 * progress exactly as before.                                        */
static void au_push(AuBuf*a,const uint8_t*pkt,int is_au_boundary){
    int afc=(pkt[3]>>4)&3;if(afc==2)return;
    int off=4;if(afc==3)off+=1+(int)pkt[4];if(off>=TS_SZ)return;
    int pusi=(pkt[1]>>6)&1;
    const uint8_t*pay=pkt+off;int plen=TS_SZ-off;
    if(pusi){
        if(plen<9||pay[0]!=0||pay[1]!=0||pay[2]!=1)return;
        uint8_t f=(pay[7]>>6)&3;
        if(f&&plen>=14){
            int64_t p=(int64_t)(((uint64_t)(pay[9]&0x0E)<<29)|
                ((uint64_t)pay[10]<<22)|((uint64_t)(pay[11]&0xFE)<<14)|
                ((uint64_t)pay[12]<<7)|((uint64_t)(pay[13]&0xFE)>>1));
            a->pts=p;a->pts_valid=1;
            a->dts=(f==3&&plen>=19)?(int64_t)(((uint64_t)(pay[14]&0x0E)<<29)|
                ((uint64_t)pay[15]<<22)|((uint64_t)(pay[16]&0xFE)<<14)|
                ((uint64_t)pay[17]<<7)|((uint64_t)(pay[18]&0xFE)>>1)):p;
        }
        int skip=9+pay[8];pay+=skip;plen-=skip;
        if(plen<=0){
            if(is_au_boundary)a->len=0;
            return;
        }
        if(is_au_boundary)a->len=0; /* start fresh AU */
        /* else: fall through and APPEND this PES packet's payload
         * onto the AU already accumulated -- do not touch a->len.   */
    }else{
        if(a->len==0)return; /* no PUSI yet — discard */
    }
    if(plen>0){
        int cp=plen;
        if(a->len+cp>AU_MAX){
            /* Would overflow -- mark this AU as poisoned. Still cap
             * the copy to whatever room is actually left (never write
             * past the buffer), but the bytes copied past this point
             * are dead data: vidxc_push will discard the whole AU
             * because of the truncated flag, never decode any of it,
             * partial or not. See AuBuf.truncated doc for rationale.  */
            cp=AU_MAX-a->len;
            a->truncated=1;
        }
        if(cp>0){memcpy(a->buf+a->len,pay,cp);a->len+=cp;}
    }
}

typedef struct{
    int             active;
    AVCodecContext *dec_ctx;
    AVPacket       *avpkt;
    AVFrame        *frame;
    AVCodecContext *enc_ctx;
    AVPacket       *enc_pkt;
    AuBuf           au;
    uint8_t        *out;
    int             out_len;
    uint8_t         vid_cc;
    int64_t         out_pts;
    int64_t         last_decoded_pts; /* most recent v->frame->pts seen
        from the decoder - used to detect a decoder that keeps emitting
        SOMETHING (so the existing zero-frame stall check never fires)
        but whose actual content has stopped advancing, a real, distinct
        failure mode confirmed via direct analysis of real production
        segments (see stall_calls doc in vidxc_push for full detail)    */
    int             last_decoded_pts_valid;
    int             stuck_pts_calls; /* consecutive frames decoded with
        an unchanged (or non-advancing) PTS - see vidxc_push           */
    float           crf;
    int             fpn,fpd;
    int             fps_confirmed; /* 0 = fpn/fpd is still just the
        vidxc_init default/placeholder, don't open the encoder on it yet
        (see vidxc_push); 1 = the real detected fps is in fpn/fpd, safe
        to open. Set at vidxc_init time if fps was already known then,
        or by fps_probe_finalize_apply once the probe completes. This
        replaces an earlier approach that let the encoder open on
        whatever fps was available and tore it down + reopened it if a
        real mismatch showed up later -- confirmed correct but wasteful
        (a real 1920x1080 50fps source decoded frame1 fast enough to
        open the encoder at the 25/1 default before the probe finished,
        needing a full teardown+reopen). Waiting to open the encoder
        specifically -- not the decoder, which is what actually caused
        the original freeze bug -- avoids ever needing that teardown:
        decoded frames just get dropped (harmless, cheap) for the small
        bounded window until fps is confirmed, then encoding starts
        with the correct rate the first time, every time.               */
    int             dec_ready;     /* gate: wait for SPS before decoding */
    int             stall_calls;   /* consecutive decode calls with zero
                                       frame output - see the dec_ready
                                       reset logic in vidxc_push for why
                                       this matters and how it's used   */
    int             force_idr;     /* force next encoder output to be a
                                       fresh IDR (used after CC recovery -
                                       see vidxc_push for full rationale) */
    int             got_keyframe;  /* set when encoder outputs an IDR */
    int             enc_started;   /* 1 after first IDR from encoder */
    int             corrupt_errors_since_check; /* incremented by the
        global av_log callback (see install_corruption_log_callback)
        whenever libavcodec logs a real reference-frame corruption
        message ("mmco: unref short failure", "reference picture
        missing during reorder") while THIS context is the one
        actively decoding. Read-and-reset by vidxc_push right after
        each decode call. This is the real, direct signal — not an
        inferred heuristic — for the bug confirmed via real production
        evidence: a single CC drop can leave the decoder in a corrupted
        reference-frame state that does NOT self-heal, with errors
        escalating across many subsequent segments despite zero
        further CC drops, until a fresh I-slice forces re-validation.*/
    int             corrupt_calls;  /* consecutive decode calls (NOT
        consecutive frames - granularity matches stall_calls/
        stuck_pts_calls for consistency) where corrupt_errors_since_check
        was nonzero - see vidxc_push for the actual threshold/reset.  */
    time_t          last_islice_log; /* wall-clock second of the last
        "I-slice found" print -- see that printf's own comment for why
        this exists (rate-limiting a potentially hot-path log line).  */
    /* Full-range -> limited-range color conversion (green/color-shift
     * fix). Some sources signal full-range (0-255, "PC"/JPEG range)
     * YUV via SPS VUI - ffprobe shows this as e.g. "yuvj420p(pc,
     * bt709,...)". The old code did a raw av_frame_copy() straight
     * from that full-range decoded frame into an output frame merely
     * LABELED yuv420p (standard limited/TV range, 16-235), with zero
     * pixel-value rescaling: the exact same numeric YCbCr values then
     * got reinterpreted one range step differently by every
     * downstream player, which is a textbook cause of a greenish/
     * color-shifted picture. range_sws (created lazily, once per
     * resolution) does the real full->limited rescale via sws_scale
     * before encoding, so every player renders correctly regardless
     * of whether it even honors an explicit full-range VUI flag
     * (embedded LG/Samsung decoders in particular are unreliable
     * about that) -- see the source-range check in vidxc_push.        */
    struct SwsContext *range_sws;
    int                range_sws_w,range_sws_h;
    enum AVPixelFormat range_sws_srcfmt;
    /* PTS sanity/wrap-normalize state (see the clamp in vidxc_push
     * right after src_pts is computed from frame->pts, for the full
     * root-cause writeup). Confirmed real, reproducible bug: three
     * separate production segments (6s apart) each showed a decoded
     * frame's pts equal to the CORRECT value plus exactly 8589934592
     * (2^33, the MPEG PTS wrap boundary at 90kHz) — to sub-millisecond
     * precision across all three occurrences — while every other
     * segment's pts was correct. Since au_push's raw 33-bit PTS
     * extraction has no wrap-add logic of its own (confirmed: the
     * only PTS math in this file is a straight bit-field read), this
     * is coming from inside libavcodec's own decode/reorder path.     */
    int64_t last_good_pts;
    int     last_good_pts_valid;
    uint8_t src_stream_type; /* the PMT stream_type this decoder was
        opened for (0x01/0x02=MPEG2, 0x1B=H.264, 0x24=HEVC) -- set once
        in vidxc_init. FIX: the I-slice recovery gate (dec_ready==0
        case in vidxc_push) used to unconditionally parse incoming
        bytes as H.264 NAL/Exp-Golomb slice syntax, regardless of what
        the actual source codec was. For an MPEG2 source, that's
        parsing the wrong grammar entirely -- MPEG2 also uses 0x000001
        start codes (shared legacy convention with H.264) but the
        byte(s) that follow mean something completely different, so
        the H.264 parse produces essentially coincidental, meaningless
        pass/fail results. Confirmed via real production log: an MPEG2
        source (STAR SPORTS 2 TELUGU) showing "[vidxc] I-slice found"
        firing repeatedly, sometimes 8-10 times within one 10-second
        window, for over an hour straight, with segment durations
        swinging wildly (1.8s to 15+s) -- the decoder could never stay
        locked on, because the gate meant to confirm "safe to resume"
        was checking for a syntax structure that doesn't exist in this
        stream. This field lets that gate branch to genuine MPEG2
        picture_coding_type parsing instead when appropriate.          */
    uint8_t mpeg2_seqhdr[256]; int mpeg2_seqhdr_len; /* BUG FIX: was 32
        bytes, sized only for a bare sequence_header+sequence_extension
        -- confirmed via real production log (send_packet failing on
        the exact AU this prepend touches, hex dump showing a
        sequence_header start code) that this source's real sequence
        header is longer than 32 bytes, almost certainly because it
        carries custom quantization matrices (each up to 64 bytes,
        optionally two of them). The fixed-size cache was silently
        truncating it mid-structure -- corrupting the exact thing this
        mechanism exists to fix. 256 bytes comfortably covers the
        worst case (sequence_header + sequence_extension + two full
        custom quantization matrices, ~150 bytes) with margin.         */
    int mpeg2_need_seqhdr_prepend; /* set for exactly one AU -- the one
        that triggers a dec_ready 0->1 transition -- then consumed and
        cleared immediately when that AU is actually sent.             */ /* cached
        sequence_header (+sequence_extension, if present) bytes for
        MPEG2 sources -- captured from whichever AU first carries one,
        and prepended to every I-frame AU sent to the decoder from then
        on (see the found_islice handling below). Mirrors the existing
        H.264 SPS/PPS caching pattern elsewhere in this codebase, for
        exactly the same underlying reason: avcodec_flush_buffers()
        (called every time this decoder resets, which happens often on
        a source needing frequent recovery) wipes previously-parsed
        sequence-level state along with everything else, so relying on
        the decoder having "already seen" a sequence_header once,
        earlier in the stream, isn't reliable -- it needs one fresh
        alongside every I-frame it's asked to decode after a reset,
        not just the very first one. 32 bytes comfortably covers a
        standard sequence_header (12 bytes) plus sequence_extension
        (10 bytes, its own 00 00 01 B5-prefixed unit).                 */
    int mpeg2_frames_since_flush; /* DIAGNOSTIC: incremented on every
        successfully decoded frame, reset to 0 at every
        avcodec_flush_buffers() call. Added to test a specific theory:
        that send_packet failures on this MPEG2 source are B-frames
        failing because a reference frame they need was wiped by a
        flush that happened for an unrelated reason shortly before.
        If failures cluster at low values here (0, 1, 2 frames since
        the last flush), that confirms the reference-wipe theory
        directly. If they're scattered evenly regardless of how long
        it's been since the last flush, that rules it out and points
        elsewhere.                                                     */
    struct timespec mpeg2_last_flush_time; int mpeg2_last_flush_time_valid;
    int mpeg2_dump_next_au; /* DIAGNOSTIC: set right after a send_packet
        failure so the very next AU handed to vidxc_push (whatever it
        turns out to be -- decoder-bound or gated out) also gets its
        full bytes dumped, not just a 32-byte prefix. Lets us see
        whether the bytes "missing" from a short failing AU actually
        show up misplaced at the front of the following AU (proof of
        a boundary-miscount bug) or never appear anywhere (proof of
        real data loss upstream of au_push).                          */
    int mpeg2_dump_seq; /* DIAGNOSTIC: monotonically increasing counter
        used to build unique dump filenames per channel per failure.   */
        /* wall-clock companion to the frame counter above -- lets the
         * failure log show both "N frames since flush" and "N
         * milliseconds since flush", since a source with variable
         * frame timing could have those disagree in an informative
         * way (e.g. many frames decoded very quickly right after a
         * flush, all within the reference window, vs. genuinely
         * spaced out over real time).                                 */
}VidXc;
/* corruption_log_callback moved below (after Ch/g_ch/g_nch are declared) —
   see the definition right after `static Ch g_ch[MAX_CHANNELS];`.
   FIX (long-run video-loss bug): the previous version of this file used a
   single global `g_corruption_target` (set just before
   avcodec_receive_frame, cleared right after) to decide which channel a
   corruption log line belonged to. On a box running multiple
   video_transcode channels across multiple NIC threads, those sub-windows
   can and do overlap — a real corruption event on channel A can get
   attributed to whichever channel B happened to also be mid-decode at
   that instant. If that keeps happening to the same channel, its own
   corrupt_errors_since_check counter never reaches the recovery
   threshold, dec_ready never resets, and that channel's video stays
   corrupted (frozen PTS, "mmco: unref short failure" / reference-frame
   errors) indefinitely — confirmed via real production evidence: video
   fully lost while audio kept working normally on the same channel,
   over a long run on a multi-channel box. Fixed by using the `avcl`
   parameter libavcodec already provides (the AVCodecContext* that
   generated the log line) to match directly against each channel's own
   dec_ctx, instead of trusting shared mutable state.                  */

static int vidxc_init(VidXc*v,uint8_t st,int fpn,int fpd,float crf){
    memset(v,0,sizeof*v);
    v->src_stream_type=st;
    enum AVCodecID cid=st2codec(st);
    if(cid==AV_CODEC_ID_NONE){
        fprintf(stderr,"[vidxc] unknown stream_type 0x%02X\n",st);return 0;}
    const AVCodec*dec=avcodec_find_decoder(cid);
    if(!dec){fprintf(stderr,"[vidxc] no decoder\n");return 0;}
    v->dec_ctx=avcodec_alloc_context3(dec);
    /* NOTE: deliberately NOT setting AV_CODEC_FLAG_LOW_DELAY here.
     * That flag tells the decoder "this stream has no B-frame reorder
     * delay, emit every frame the instant it's decoded" -- which
     * skips libavcodec's internal reorder buffer entirely. Confirmed
     * via a decrypted real output segment that this source DOES carry
     * B-frames: with the flag set, avcodec_receive_frame handed us
     * frames back in raw DECODE order (I,P,B,B,P,B,B,...) instead of
     * DISPLAY order, each still carrying its own correct PTS -- and
     * since our encoder is configured with max_b_frames=0 (re-stamps
     * whatever order it receives, no reordering of its own), that
     * wrong decode-order sequence went straight into the output H.264
     * stream. PTS values looked fine in isolation but jumped around
     * (-80ms/+40ms/+160ms, repeating every 3 frames -- the exact
     * signature of an IbbP GOP being read off in decode order), and
     * any player showing frames in file order displays visibly
     * out-of-order motion -- the "frame by frame" judder reported in
     * production. Removing the flag lets the decoder buffer and
     * reorder B-frames internally as it's designed to, exactly like
     * every other MPEG2 decode path that doesn't force low-delay.    */
    v->dec_ctx->thread_count=1;
    if(avcodec_open2(v->dec_ctx,dec,NULL)<0){
        avcodec_free_context(&v->dec_ctx);return 0;}
    v->avpkt=av_packet_alloc();
    v->frame=av_frame_alloc();
    v->enc_pkt=av_packet_alloc();
    v->out=malloc(VOUT_MAX*TS_SZ);
    v->crf=crf>0?crf:23.0f;
    v->fpn=fpn?fpn:25;v->fpd=fpd?fpd:1;
    v->fps_confirmed=(fpn>0&&fpd>0); /* real value passed in -> already
        confirmed, safe to open the encoder on it right away (this is
        the normal case if the fps probe happened to finish before the
        PMT parse that creates this VidXc); 0/0 -> placeholder default,
        encoder waits until fps_probe_finalize_apply confirms it later. */
    au_reset(&v->au); v->active=1;
    printf("[vidxc] decoder: %s fps=%d/%d crf=%.0f\n",dec->name,fpn,fpd,crf);
    return 1;
}
static int vidxc_open_enc(VidXc*v,int w,int h,AVRational sar,
                           enum AVColorSpace cspace,enum AVColorPrimaries cprim,
                           enum AVColorTransferCharacteristic ctrc){
    /* Try encoders in order: OpenH264 -> libx264 -> any H264 */
    const AVCodec*enc=avcodec_find_encoder_by_name("libopenh264");
    if(!enc) enc=avcodec_find_encoder_by_name("libx264");
    if(!enc) enc=avcodec_find_encoder(AV_CODEC_ID_H264);
    if(!enc){fprintf(stderr,"[vidxc] no H264 encoder found\n");return 0;}

    v->enc_ctx=avcodec_alloc_context3(enc);
    v->enc_ctx->width=w; v->enc_ctx->height=h;
    if(sar.num>0&&sar.den>0)v->enc_ctx->sample_aspect_ratio=sar;
    v->enc_ctx->time_base=(AVRational){v->fpd,v->fpn};
    v->enc_ctx->framerate=(AVRational){v->fpn,v->fpd};
    v->enc_ctx->gop_size=v->fpn*2/v->fpd; /* IDR every 2s */
    v->enc_ctx->max_b_frames=0;            /* no B-frames: DTS=PTS */
    v->enc_ctx->pix_fmt=AV_PIX_FMT_YUV420P;
    /* Propagate the SOURCE's real color matrix/primaries/transfer
     * (e.g. bt709 for HD, confirmed via ffprobe on real production
     * streams) so libx264 writes correct VUI instead of leaving it
     * unspecified — an unspecified/wrong matrix is the other classic
     * half of the greenish/color-shifted-picture bug (the range
     * mismatch, fixed in vidxc_push via range_sws, is the other half).
     * color_range is ALWAYS forced to MPEG (limited) here, regardless
     * of the source's range, because vidxc_push always rescales
     * full-range source frames down to limited range before they ever
     * reach the encoder — so limited range is what the encoder is
     * actually receiving, and what must be signalled.
     * NOTE: libopenh264 (tried first, above) does not forward these
     * fields into its output VUI at all (a real ffmpeg-wrapper
     * limitation, not something fixable from this side) — this only
     * takes visible effect on streams that fall back to libx264. The
     * range fix in vidxc_push, however, applies with either encoder.  */
    v->enc_ctx->color_range=AVCOL_RANGE_MPEG;
    v->enc_ctx->colorspace=cspace;
    v->enc_ctx->color_primaries=cprim;
    v->enc_ctx->color_trc=ctrc;
    v->enc_ctx->thread_count=0; /* Single-threaded by design for high
                                    channel-count deployments (e.g. ~300
                                    concurrent streams on shared hardware).
                                    Multithreading (thread_count=0, auto)
                                    was tested and reverted: it raised
                                    per-channel CPU from ~100% to ~150-160%
                                    (real measurement) for a stream that
                                    was already real-time at 100%, and at
                                    high channel counts also risks thread
                                    oversubscription — auto-detect sizes
                                    threads off the WHOLE machine's core
                                    count, not accounting for the other
                                    ~300 channel processes competing for
                                    those same cores, which causes
                                    scheduler thrashing rather than real
                                    speedup. Fixed cost-per-channel matters
                                    more than per-channel speed at this
                                    scale. One packet per frame is also a
                                    nice side benefit (see sliced_threads
                                    below for the setting that actually
                                    matters for that guarantee).
                                    NOTE: this field was left at 0 (auto)
                                    despite this exact comment saying that
                                    was reverted — auto-sizing off the
                                    whole machine's core count with no
                                    awareness of ~300 sibling channel
                                    processes doing the same thing is the
                                    single most likely cause of large
                                    (5-15%, vs an expected 1-3%) run-to-run
                                    CPU variance per channel: which channel
                                    "wins" contested cores at any instant
                                    is scheduler luck, not fixed cost.     */

    if(strcmp(enc->name,"libopenh264")==0){
        /* OpenH264: ABR mode, map CRF quality to bitrate.
         * crf=18~4Mbps, crf=23~2.5Mbps, crf=28~1.5Mbps, crf=32~900kbps.
         * Decay constant verified against this exact ladder: with the
         * old 0.82f, actual output at crf=23/28/32 was only ~1.48/
         * 0.55/0.25 Mbps — 41-72% BELOW the documented/intended
         * bitrate at every CRF except the crf=18 anchor point itself
         * (where the exponent is 0, so no decay constant applies).
         * Every channel at the default crf=23 has been silently
         * encoding at barely half its intended bitrate — the most
         * likely direct cause of poor SD clarity specifically (SD's
         * lower resolution makes under-bitrating far more visible,
         * i.e. blockier/softer, than the same shortfall on HD).
         * 0.90f was solved against all three documented anchor points
         * (18->23, 23->28, 28->32) and matches each within ~7%.        */
        int br=(int)(4000000.0f*powf(0.90f,v->crf-18.0f));
        if(br<400000)br=400000; if(br>8000000)br=8000000;
        v->enc_ctx->bit_rate=br;
        v->enc_ctx->rc_max_rate=br;
        v->enc_ctx->rc_buffer_size=br*2;
        /* Main profile: supported by all LG/Samsung/VLC/ExoPlayer */
        av_opt_set(v->enc_ctx->priv_data,"profile","main",0);
        /* Always allow max quality — no frame skipping */
        av_opt_set_int(v->enc_ctx->priv_data,"allow_skip_frames",0,0);
        /* Complexity: 0=fastest, no B-frames by default in OpenH264 */
        av_opt_set_int(v->enc_ctx->priv_data,"loopfilter",1,0);
    } else if(strcmp(enc->name,"libx264")==0){
        /* Resolution-aware preset: "ultrafast" (the WORST-quality x264
         * preset — minimal motion search, no trellis, no subpel
         * refinement) was hardcoded for every channel regardless of
         * resolution. That's defensible for HD at 300-channel scale
         * (CPU-per-channel matters more there — see the thread_count=0
         * doc above), but SD has vastly fewer macroblocks per frame:
         * a slower/better preset costs proportionally far less CPU on
         * SD than the same preset bump would on HD, so there's no
         * reason to pay full "ultrafast" quality loss on SD too. This
         * is very likely the actual cause of poor SD clarity reported
         * — since this system has no libopenh264 support at all (per
         * the real ffprobe build-config banner, no --enable-libopenh264
         * anywhere), EVERY channel has been running the libx264
         * fallback below, at "ultrafast", the whole time.
         * "veryfast" was chosen for SD (h<=576) as a safe first step:
         * meaningfully better subpel/motion search than ultrafast, at
         * a cost increase that's small in absolute terms specifically
         * because SD has so few macroblocks — bump further (faster/
         * fast/medium) if there's still CPU headroom to spend.        */
        const char*preset=(h<=576)?"veryfast":"ultrafast";
        av_opt_set(v->enc_ctx->priv_data,"preset",preset,0);
        av_opt_set(v->enc_ctx->priv_data,"tune","zerolatency",0);
        /* tune=zerolatency disables CABAC by default, which silently
         * forces x264 into Constrained Baseline profile regardless of
         * any profile setting — re-enable cabac=1 explicitly to get
         * true Main profile output (required by many TV/STB decoders
         * that reject Baseline streams or behave inconsistently with
         * them). ctx->profile is the correct way to request the
         * profile; "profile=main" inside x264-params is NOT a valid
         * key for libx264's param parser and is silently ignored.   */
        v->enc_ctx->profile=FF_PROFILE_H264_MAIN;
        /* FIX: force real headroom above the H.264 Annex A Level
         * ceiling instead of leaving level unset (x264 then picks the
         * tightest level that nominally fits, which for SD/25fps lands
         * EXACTLY on Level 3.0's ceiling: 720x576 = 1620 macroblocks,
         * needing exactly 1620*25 = 40500 MB/s against Level 3.0's
         * MaxMBPS of exactly 40500 -- zero headroom. Re-confirmed via
         * this exact channel (MPEG2-sourced, video_transcode): every
         * single segment checked landed on 0.0% headroom at Level 3.0.
         * Lenient decoders (ExoPlayer, VLC) don't enforce Annex A
         * conformance and play it anyway; several native Smart TV HLS
         * stacks do enforce it, and PCR/encoder jitter nudging the
         * real delivered rate a hair over the nominal 25.000fps is
         * enough to push a zero-headroom stream over the strict limit.
         * Level 3.1 raises MaxMBPS to 108000 (720x576 needs only
         * 40500 -- ~62% headroom) and MaxFS to 3600 (well above 1620)
         * -- pure conformance-flag correction, no real encode-quality
         * change, since the content already fits comfortably inside
         * Level 3.1's limits. HD (h>576) doesn't hit this at the
         * resolutions this pipeline handles, but level=41 is set for
         * it too, for the same reason: cheap, safe headroom instead of
         * whatever x264 would auto-pick as the tightest fit.          */
        v->enc_ctx->level=(h<=576)?31:41;
        char p[256];
        snprintf(p,sizeof p,
            "crf=%.0f:repeat_headers=1:annexb=1:bframes=0"
            ":sliced_threads=0:cabac=1:aud=1:aq-mode=2:aq-strength=0.8"
            ":scenecut=0",
            v->crf);
        /* scenecut=0: without this, x264's scene-cut detection can
         * insert an "open" I-frame (a real intra-coded frame, but NOT
         * flagged as an IDR) at a detected scene change that lands
         * near -- but not exactly on -- a segment boundary. A fresh
         * decoder starting mid-stream on a segment whose first frame
         * is an open I-frame rather than a true IDR has no clean
         * random-access point to anchor to. Disabling scene-cut
         * detection means every I-frame is a deliberate, real IDR at
         * the gop_size boundary this code controls directly.          */
        /* aq-mode=2 (auto-variance, dark-scene biased) + aq-strength=0.8:
         * redistributes bits toward low-variance/flat regions (skin
         * tones, backgrounds, low-motion news/talk content — common on
         * SD broadcast channels) instead of spending them uniformly.
         * This is the standard perceptual-quality lever x264 has that
         * OpenH264 has NO equivalent for at all, so it only helps here
         * because libx264 is what's actually encoding on this build.
         * Cheap: aq analysis, unlike subme/trellis, doesn't meaningfully
         * add CPU cost on top of whatever preset is already chosen —
         * safe to apply at both presets above, not just the SD one.   */
        av_opt_set(v->enc_ctx->priv_data,"x264-params",p,0);
    } else {
        v->enc_ctx->bit_rate=2000000;
    }

    if(avcodec_open2(v->enc_ctx,enc,NULL)<0){
        fprintf(stderr,"[vidxc] encoder open failed (%s)\n",enc->name);
        avcodec_free_context(&v->enc_ctx); return 0;}

    printf("[vidxc] encoder: %s %dx%d %.2ffps  bitrate=%dkbps\n",
           enc->name,w,h,(float)v->fpn/v->fpd,
           (int)(v->enc_ctx->bit_rate/1000));
    return 1;
}
static void vidxc_free(VidXc*v){
    if(!v->active)return;
    avcodec_free_context(&v->enc_ctx);
    avcodec_free_context(&v->dec_ctx);
    av_frame_free(&v->frame);
    av_packet_free(&v->avpkt);
    av_packet_free(&v->enc_pkt);
    if(v->range_sws){sws_freeContext(v->range_sws);v->range_sws=NULL;}
    free(v->out);v->active=0;
}
static void vidxc_write(VidXc*v,const uint8_t*data,int size,
                         uint16_t pid,int64_t pts,int64_t dts,int disc){
    int so=0,first=1;
    while(so<size){
        if(v->out_len+TS_SZ>VOUT_MAX*TS_SZ){
            /* This should not happen with VOUT_MAX sized generously above
             * any realistic frame size, but if it ever does, truncating
             * silently here corrupts the H264 bytestream mid-slice-data
             * (confirmed: this exact path caused real "bytestream
             * overread"/macroblock decode errors on a real broadcast
             * capture when VOUT_MAX was too small). Make it loud instead
             * of silent so a future regression is immediately visible
             * rather than manifesting as mysterious intermittent
             * corruption far away from this code.                       */
            static int _warned=0;
            if(_warned<5){
                fprintf(stderr,"[vidxc] WARNING: frame truncated! "
                        "size=%d exceeds VOUT_MAX=%d packets (%d bytes) — "
                        "increase VOUT_MAX\n",size,VOUT_MAX,VOUT_MAX*TS_SZ);
                _warned++;
            }
            break;
        }
        uint8_t*pkt=v->out+v->out_len;
        memset(pkt,0,TS_SZ); /* zero, not 0xFF — payload area must never
                                 contain stray 0xFF that could be mistaken
                                 for H264 stream data by NAL scanners       */
        pkt[0]=0x47;
        pkt[1]=(uint8_t)((first?0x40:0x00)|((pid>>8)&0x1F));
        pkt[2]=(uint8_t)(pid&0xFF);
        int po;
        if(first){
            /* CRITICAL: carry PCR in the adaptation field of the first TS
             * packet of every video frame. Without PCR the player has no
             * system clock reference and stalls/freezes frame-by-frame
             * even though valid PES data is arriving.                    */
            pkt[3]=(uint8_t)(0x30|(v->vid_cc&0x0F)); /* AFC=11: adapt+payload */
            v->vid_cc=(v->vid_cc+1)&0x0F;
            pkt[4]=7;            /* adaptation_field_length */
            pkt[5]=(uint8_t)(0x10|(disc?0x80:0x00)); /* PCR_flag=1,
                discontinuity_indicator=1 if this frame follows a real
                recovery (see vidxc_write's doc above for full
                rationale — this is the actual fix for VLC-specific
                video freezes on channels that experience CC-drop
                recovery, confirmed to play fine on ExoPlayer/LG with
                the exact same unflagged bytes).                      */
            uint64_t pcr_base=(uint64_t)pts; /* 90kHz base */
            uint16_t pcr_ext=0;  /* 27MHz extension, 0 = aligned to 90kHz */
            pkt[6]=(uint8_t)((pcr_base>>25)&0xFF);
            pkt[7]=(uint8_t)((pcr_base>>17)&0xFF);
            pkt[8]=(uint8_t)((pcr_base>>9)&0xFF);
            pkt[9]=(uint8_t)((pcr_base>>1)&0xFF);
            pkt[10]=(uint8_t)(((pcr_base&1)<<7)|0x7E|((pcr_ext>>8)&1));
            pkt[11]=(uint8_t)(pcr_ext&0xFF);
            int ao=4+1+7; /* TS hdr + adapt_len byte + 7 adapt bytes */
            uint8_t*p=pkt+ao;
            p[0]=0;p[1]=0;p[2]=1;p[3]=0xE0;
            /* PES_packet_length = PES_header_data_length(10) + payload size.
             * Previously hardcoded to 0 — technically a legal "unbounded"
             * marker for video PES per spec, but our packetizer doesn't
             * actually leave the packet unbounded (it's TS-framed normally),
             * and ffmpeg's demuxer choked on the mismatch between the
             * claimed-unbounded length and the actual bounded structure,
             * logging "PES packet size mismatch" and momentarily losing
             * sync — which manifested as spurious "non-existing PPS"
             * warnings on perfectly valid SPS/PPS/IDR data. Real hardware
             * decoders (ExoPlayer, LG, Samsung) are far less forgiving of
             * this than ffmpeg's resync logic, matching the reported
             * macroblock/freeze symptoms. Field is 16 bits; per spec,
             * if the payload would overflow it, fall back to 0 (legal
             * unbounded marker) rather than write a truncated/wrong value.*/
            {
                /* PES_packet_length counts everything AFTER this 16-bit
                 * field itself: 2 flag bytes (p[6],p[7]) + 1 byte for
                 * header_data_length (p[8]) + header_data_length(10) +
                 * the actual payload. That's 2+1+10+size = 13+size.
                 * Previous code wrote 10+size — missing the 3 bytes for
                 * the flags+hdr_len_byte — which meant every single PES
                 * packet's declared length was exactly 3 bytes short of
                 * what was actually delivered. Confirmed by direct
                 * measurement against a real broadcast capture: every
                 * frame's actual payload was exactly claimed_length+3.
                 * This is why ffmpeg logged "Packet corrupt"/"PES packet
                 * size mismatch" on literally every frame in every
                 * segment, and in at least one case caused real MB/MV
                 * concealment errors (visible corruption), not just a
                 * benign warning.                                        */
                int pes_len = 13 + size;
                if(pes_len > 0xFFFF) pes_len = 0; /* spec-legal unbounded */
                p[4]=(uint8_t)((pes_len>>8)&0xFF);
                p[5]=(uint8_t)(pes_len&0xFF);
            }
            p[6]=0x80;p[7]=0xC0;p[8]=0x0A;
            p[9] =(uint8_t)(0x31|((pts>>29)&0x0E));
            p[10]=(uint8_t)((pts>>22)&0xFF);
            p[11]=(uint8_t)(0x01|((pts>>14)&0xFE));
            p[12]=(uint8_t)((pts>>7)&0xFF);
            p[13]=(uint8_t)(0x01|((pts<<1)&0xFE));
            p[14]=(uint8_t)(0x11|((dts>>29)&0x0E));
            p[15]=(uint8_t)((dts>>22)&0xFF);
            p[16]=(uint8_t)(0x01|((dts>>14)&0xFE));
            p[17]=(uint8_t)((dts>>7)&0xFF);
            p[18]=(uint8_t)(0x01|((dts<<1)&0xFE));
            po=ao+19; first=0;
        } else {
            po=4;
            int remain=size-so;
            if(remain<TS_SZ-4){
                /* LAST packet of this frame and it won't fill the TS
                 * packet exactly. Use a proper adaptation field with
                 * stuffing_byte padding (AFC=11) — this is the ONLY
                 * spec-legal way to pad a TS packet. Writing raw 0xFF
                 * directly into the payload area (old bug) corrupts the
                 * H264 byte stream: NAL scanners in hardware decoders
                 * (Android/LG/Samsung) don't respect PES packet_length
                 * and parse the stuffing as bogus trailing NAL/start-code
                 * bytes, producing macroblock corruption on every frame
                 * and confusing frame boundaries enough to stall decode
                 * pacing (the "frame by frame" freeze).                  */
                int pad=(TS_SZ-4)-remain; /* stuffing bytes needed */
                pkt[3]=(uint8_t)(0x30|(v->vid_cc&0x0F)); /* AFC=11 */
                v->vid_cc=(v->vid_cc+1)&0x0F;
                int afl=pad-1; /* adaptation_field_length itself counts
                                  as 1 of the pad bytes via its own field */
                if(afl<0)afl=0;
                pkt[4]=(uint8_t)afl;
                if(afl>0){
                    pkt[5]=0x00; /* no flags set, pure stuffing follows */
                    for(int i=1;i<afl;i++)pkt[5+i]=0xFF; /* stuffing_byte */
                }
                po=4+1+afl; /* TS hdr + adapt_len byte + adapt field body */
            } else {
                pkt[3]=(uint8_t)(0x10|(v->vid_cc&0x0F)); /* AFC=01: full payload */
                v->vid_cc=(v->vid_cc+1)&0x0F;
            }
        }
        int sp=TS_SZ-po,cp=size-so;if(cp>sp)cp=sp;
        memcpy(pkt+po,data+so,cp);so+=cp;v->out_len+=TS_SZ;
    }
}
/* Returns number of output TS packets produced */
static int vidxc_push(VidXc*v,const uint8_t*ts_pkt,uint16_t pid){
    v->out_len=0;
    int pusi=(ts_pkt[1]>>6)&1;
    int is_mpeg2=(v->src_stream_type==0x01||v->src_stream_type==0x02);
    /* Whether this PUSI genuinely starts a new access unit, or is
     * just the next in a run of PES packets this muxer splits one
     * picture's elementary-stream data across (see mpeg2_is_au_start
     * doc). MUST be computed the same way here (the flush/dump/decode
     * gate) and inside au_push (the accumulate gate) -- both derive
     * it from the same ts_pes_es_payload()+mpeg2_is_au_start() check,
     * passed down as a single precomputed value, so they can't
     * disagree with each other the way the previous, incomplete fix
     * did (au_push correctly appended, but this flush gate below
     * still fired -- and dumped/decoded -- on every raw PUSI
     * regardless).                                                    */
    int is_au_boundary=0;
    if(pusi){
        if(!is_mpeg2){
            is_au_boundary=1;
        }else{
            const uint8_t*pay;int plen;
            if(ts_pes_es_payload(ts_pkt,&pay,&plen))
                is_au_boundary=(v->au.len==0)||mpeg2_is_au_start(pay,plen);
            else
                is_au_boundary=0; /* Unparseable PES-start packet: au_push
                    will independently re-validate and silently drop this
                    exact packet on its own (its own inline check, same
                    condition) regardless of what we pass it -- it never
                    resets or appends on a packet it can't parse. This
                    flush gate must agree: forcing a boundary here (the
                    previous version of this fix did exactly that) would
                    make US treat the in-progress AU as "complete" and
                    decode/dump it right now, while au_push considers it
                    still open -- exactly the kind of gate/accumulate
                    mismatch that caused the original bug. Doing nothing
                    here just leaves the AU accumulating, waiting for the
                    next packet that actually parses.                    */
        }
    }
    if(is_au_boundary&&v->au.len>0){
        /* DIAGNOSTIC: dump this AU's raw, as-assembled bytes (before
         * any gating/truncation handling below) whenever we're inside
         * a post-failure capture window. This fires unconditionally on
         * every complete AU while the window is open -- unlike the
         * send_packet-site dump, it doesn't care whether this AU ever
         * reaches the decoder (a failure resets dec_ready and forces
         * I-slice re-validation, so the very next AU after a failure
         * usually does NOT reach send_packet -- it has to clear the
         * gate first). Comparing consecutive dumps tells us whether
         * the bytes missing from a short failing AU show up misplaced
         * at the front of the next one (boundary-miscount bug) or
         * never appear anywhere (real upstream data loss).            */
        if(v->mpeg2_dump_next_au>0){
            char _dp[160];
            snprintf(_dp,sizeof _dp,"/tmp/au_dump_ch%p_%03d.bin",
                     (void*)v,v->mpeg2_dump_seq++);
            FILE*_df=fopen(_dp,"wb");
            if(_df){fwrite(v->au.buf,1,(size_t)v->au.len,_df);fclose(_df);
                fprintf(stderr,"[vidxc] DIAGNOSTIC dumped %d-byte AU -> %s "
                        "(capture window: %d more after this)\n",
                        v->au.len,_dp,v->mpeg2_dump_next_au-1);}
            v->mpeg2_dump_next_au--;
        }
        /* A truncated AU (see AuBuf.truncated doc) must never reach
         * the decoder, partial or not -- a real H264 AU cut off
         * mid-slice-data (not at a NAL boundary) can desync the
         * entropy decoder or corrupt SPS/PPS state for the picture.
         * Discard it entirely and force I-slice re-validation on the
         * next AU, exactly like the existing CC-drop/decode-error
         * recovery path already does.                                 */
        if(v->au.truncated){
            static int _trunc_warn=0;
            if(_trunc_warn++<5)
                fprintf(stderr,"[vidxc] WARNING: discarding truncated AU "
                        "(%d bytes, exceeded AU_MAX) -- forcing I-slice "
                        "re-validation\n",v->au.len);
            v->dec_ready=0;au_reset(&v->au);
            v->stuck_pts_calls=0;v->last_decoded_pts_valid=0;
            goto push_done;
        }
        /* Gate: only send AU to decoder once we've seen a genuine
         * I-frame/I-slice. Without this, decoder gets P-frames first
         * -> INVALIDDATA.
         * FIX: this used to unconditionally parse as H.264 NAL/Exp-
         * Golomb slice syntax regardless of the actual source codec.
         * For an MPEG2 source that's the wrong grammar entirely --
         * MPEG2 shares the 0x000001 start-code convention but the
         * following bytes mean something completely different, so the
         * H.264 parse produced coincidental, unreliable results
         * (confirmed via real production log: an MPEG2 channel never
         * staying decode-locked, "[vidxc] I-slice found" firing
         * repeatedly for over an hour straight). Branch to the correct
         * grammar for the actual source codec instead.                */
        if(!v->dec_ready){
            const uint8_t*d=v->au.buf;int sz=v->au.len;
            int found_islice=0;
            if(v->src_stream_type==0x01||v->src_stream_type==0x02){
                /* MPEG2: real picture_coding_type check, no NAL/Exp-
                 * Golomb concepts involved at all — see is_mpeg2_iframe
                 * doc for the exact bit layout.                        */
                found_islice=is_mpeg2_iframe(d,sz);
            } else {
            for(int i=0;i+3<sz;i++){
                int nt=-1;int hdr_off=-1;
                if(d[i]==0&&d[i+1]==0&&d[i+2]==0&&i+4<sz&&d[i+3]==1){nt=d[i+4]&0x1F;hdr_off=i+4;}
                else if(d[i]==0&&d[i+1]==0&&d[i+2]==1){nt=d[i+3]&0x1F;hdr_off=i+3;}
                else continue;
                if(nt==1||nt==5){
                    /* found a slice NAL — check if it's really an
                     * I-slice, regardless of whether nt==5 (IDR) or
                     * nt==1 (this source uses non-IDR NALs for what
                     * are functionally I-slices, a known real quirk
                     * of this broadcast encoder — see is_islice doc) */
                    int payload_off=hdr_off+1;
                    if(is_islice(d+payload_off,sz-payload_off)){found_islice=1;}
                    break; /* only the first slice NAL matters — an AU
                              has one picture, one slice_type for our
                              purposes (multi-slice pictures all share
                              the same type in practice for this gate) */
                }
            }
            }
            if(!found_islice){
                /* Even when this AU isn't itself a usable I-frame, it
                 * may carry (or be, on its own) a sequence_header —
                 * cache it unconditionally so it's available to prepend
                 * whenever a real I-frame does arrive. See
                 * mpeg2_seqhdr field doc for why this needs to be kept
                 * fresh across resets, not just captured once.         */
                if((v->src_stream_type==0x01||v->src_stream_type==0x02)&&
                   has_mpeg2_seq_header(d,sz)&&sz<=(int)sizeof(v->mpeg2_seqhdr)){
                    memcpy(v->mpeg2_seqhdr,d,sz);v->mpeg2_seqhdr_len=sz;
                }
                au_reset(&v->au);goto push_done;
            }
            /* Found a genuine I-slice — flush decoder of any stale
             * reference frames, then let this AU through.
             * Rate-limited: this fires every time dec_ready transitions
             * 0->1, which on a noisy source (frequent CC drops, or
             * frequent truncated-AU resyncs) could otherwise print many
             * times per second per channel -- a real, measurable cost
             * at high channel counts (stdout lock contention, syscall
             * overhead). Cap at once per second per channel; the
             * SIGUSR1 dump and stats line already give visibility into
             * dec_ready's current state without needing every single
             * transition logged.                                       */
            avcodec_flush_buffers(v->dec_ctx);
            v->dec_ready=1;
            v->mpeg2_frames_since_flush=0; /* DIAGNOSTIC reset -- see
                field doc for what this is testing.                    */
            clock_gettime(CLOCK_MONOTONIC,&v->mpeg2_last_flush_time);
            v->mpeg2_last_flush_time_valid=1;
            v->stuck_pts_calls=0;v->last_decoded_pts_valid=0;
            v->last_good_pts_valid=0; /* fresh decode restart — don't
                let the wrap/sanity clamp compare against a pts from
                before this discontinuity.                            */
            v->mpeg2_need_seqhdr_prepend=
                (v->src_stream_type==0x01||v->src_stream_type==0x02)&&
                v->mpeg2_seqhdr_len>0&&!has_mpeg2_seq_header(d,sz);
            /* ^ avcodec_flush_buffers() just wiped any previously-
             * parsed sequence-level state (dimensions, chroma format,
             * etc.) along with everything else, so this I-frame AU
             * needs a fresh sequence_header alongside it -- UNLESS it
             * already carries its own (some sources do bundle both in
             * one AU, in which case prepending the cached one too
             * would just be redundant, harmless but wasteful).         */
            {time_t now_t=time(NULL);
             if(now_t!=v->last_islice_log){
                 printf("[vidxc] I-slice found — decoder flushed and ready\n");
                 v->last_islice_log=now_t;}}
        }
        uint8_t*combined_mpeg2=NULL;
        if(v->mpeg2_need_seqhdr_prepend){
            v->mpeg2_need_seqhdr_prepend=0;
            combined_mpeg2=malloc(v->mpeg2_seqhdr_len+v->au.len);
            if(combined_mpeg2){
                memcpy(combined_mpeg2,v->mpeg2_seqhdr,v->mpeg2_seqhdr_len);
                memcpy(combined_mpeg2+v->mpeg2_seqhdr_len,v->au.buf,v->au.len);
                v->avpkt->data=combined_mpeg2;
                v->avpkt->size=v->mpeg2_seqhdr_len+v->au.len;
            } else {
        v->avpkt->data=v->au.buf;v->avpkt->size=v->au.len;
            }
        } else {
            v->avpkt->data=v->au.buf;v->avpkt->size=v->au.len;
        }
        v->avpkt->pts=v->au.pts_valid?v->au.pts:AV_NOPTS_VALUE;
        v->avpkt->dts=v->au.pts_valid?v->au.dts:AV_NOPTS_VALUE;
        {
            /* MPEG2 defensive guard: an AU can reach this point with
             * sequence_header/group_start/picture_header start codes
             * present (is_mpeg2_iframe above only requires a picture
             * header with coding_type==I -- it never checks for actual
             * slice data) but NO slice_start_code (0x01-0xAF) anywhere
             * in it -- i.e. headers with nothing to decode. That's
             * exactly what a muxer that puts these headers in their
             * own PES packet, separate from the picture data that
             * follows in a later PES packet, would produce for the
             * FIRST such packet. avcodec_send_packet is certain to
             * reject this (there's no picture to decode), so don't
             * even call it -- that would burn a dec_ready reset on
             * something that isn't corruption, just an AU that's
             * legitimately still waiting on its own slice data.        */
            int _skip_headers_only=0;
            if(v->src_stream_type==0x01||v->src_stream_type==0x02){
                int _has_slice=0;
                for(int _si=0;_si+3<v->avpkt->size;_si++){
                    if(v->avpkt->data[_si]==0&&v->avpkt->data[_si+1]==0&&
                       v->avpkt->data[_si+2]==1&&v->avpkt->data[_si+3]>=0x01&&
                       v->avpkt->data[_si+3]<=0xAF){_has_slice=1;break;}
                }
                _skip_headers_only=!_has_slice;
            }
            if(_skip_headers_only){
                static int _hdronly_log=0;
                if(_hdronly_log++<10)
                    fprintf(stderr,"[vidxc] MPEG2 AU has header start "
                            "codes but no slice data (%d bytes) -- "
                            "skipping decode, not counted as a failure\n",
                            v->avpkt->size);
                if(combined_mpeg2)free(combined_mpeg2);
                goto push_done;
            }
            /* corruption_log_callback now attributes errors via avcl
               (matched against v->dec_ctx) instead of a set/clear
               window around this call — see that function's comment. */
            int _sp=avcodec_send_packet(v->dec_ctx,v->avpkt);
            /* DEBUG: capture a hex prefix of what was actually sent,
             * BEFORE freeing combined_mpeg2 -- v->avpkt->data may point
             * into that buffer, so this must happen before the free
             * below, not after (reading freed memory would be a real
             * use-after-free bug, not just a logging inconvenience).
             * Added specifically to find the real cause of the
             * remaining, unexplained send_packet failures on MPEG2
             * sources: a raw capture of the actual source confirmed
             * every genuine access unit in this content is >=4497
             * bytes, but the AUs failing here are 157-2002 bytes --
             * nothing that small exists in the real content, so
             * something is truncating data before it reaches this
             * point. This dump shows exactly what's really in the
             * truncated AU (still-valid-looking MPEG2 prefix? garbage?
             * zero-filled?) instead of continuing to infer indirectly.
             * Cheap enough to compute unconditionally (32 bytes, one
             * snprintf) rather than conditionally only on error, since
             * conditioning it would still need to happen before the
             * free either way.                                        */
            char _hexdump[3*32+1]="";
            int _pic_type=-1; /* DIAGNOSTIC: 1=I 2=P 3=B, -1=not found */
            {
                int _hn=v->avpkt->size<32?v->avpkt->size:32;
                for(int _hi=0;_hi<_hn;_hi++)
                    snprintf(_hexdump+_hi*3,4,"%02x ",v->avpkt->data[_hi]);
                if(v->src_stream_type==0x01||v->src_stream_type==0x02){
                    for(int _pi=0;_pi+5<v->avpkt->size;_pi++){
                        if(v->avpkt->data[_pi]==0&&v->avpkt->data[_pi+1]==0&&
                           v->avpkt->data[_pi+2]==1&&v->avpkt->data[_pi+3]==0x00){
                            uint16_t _h16=((uint16_t)v->avpkt->data[_pi+4]<<8)|
                                          v->avpkt->data[_pi+5];
                            _pic_type=(_h16>>3)&0x7;
                            break;
                        }
                    }
                }
            }
            double _ms_since_flush=-1;
            if(v->mpeg2_last_flush_time_valid){
                struct timespec _now;clock_gettime(CLOCK_MONOTONIC,&_now);
                _ms_since_flush=(_now.tv_sec-v->mpeg2_last_flush_time.tv_sec)*1000.0+
                                 (_now.tv_nsec-v->mpeg2_last_flush_time.tv_nsec)/1e6;
            }
            /* DIAGNOSTIC: full dump of the exact bytes handed to
             * avcodec_send_packet on failure -- must happen here,
             * before combined_mpeg2 is freed below, for the same
             * use-after-free reason the 32-byte hex prefix above does.
             * Note this may be v->au.buf+prepended-seqhdr, not just
             * v->au.buf, when mpeg2_need_seqhdr_prepend fired -- that's
             * intentional, it's the literal bytes the decoder rejected. */
            char _dumppath[160]="";
            if(_sp!=0){
                static int _dumpseq=0;
                snprintf(_dumppath,sizeof _dumppath,
                         "/tmp/au_fail_ch%p_%03d.bin",(void*)v,_dumpseq++);
                FILE*_df=fopen(_dumppath,"wb");
                if(_df){fwrite(v->avpkt->data,1,(size_t)v->avpkt->size,_df);
                    fclose(_df);}
                v->mpeg2_dump_next_au=3; /* open capture window: dump the
                    next 3 complete AUs (see the pusi-flush dump above),
                    regardless of whether they clear the I-slice gate.  */
            }
            if(combined_mpeg2)free(combined_mpeg2);
            static int _dec_log=0;
            if(!_dec_log&&_sp==0){printf("[vidxc] first send_packet OK len=%d\n",v->avpkt->size);_dec_log=1;}
            if(_sp!=0){
                static int _se=0;
                if(_se++<20)
                    printf("[vidxc] send_packet err=%d len=%d pic_type=%d "
                           "(1=I 2=P 3=B) frames_since_flush=%d "
                           "ms_since_flush=%.1f dump=%s hex[0:32]=%s\n",
                           _sp,v->avpkt->size,_pic_type,
                           v->mpeg2_frames_since_flush,_ms_since_flush,
                           _dumppath,_hexdump);
                /* CRITICAL: a decode failure must not be silently
                 * tolerated forever. dec_ready only gets reset to 0 on
                 * a CC drop — without this, a single bad AU right
                 * after a recovery (plausible: CC drops mean lost
                 * packets, and the AU that triggered dec_ready=1 can
                 * itself still be missing trailing data from that same
                 * loss event) would leave dec_ready stuck at 1
                 * forever, permanently skipping I-slice re-validation
                 * for every subsequent AU — even perfectly healthy
                 * ones — with no path back to a working decoder state
                 * until another CC drop happens to occur. Confirmed
                 * directly against two real production logs: exactly
                 * this sequence (CC drop -> one decode attempt -> dead
                 * silence) left the segment-cutting pipeline frozen
                 * (segs= never incremented again) for 30+ minutes of
                 * otherwise-healthy packet flow, in both cases.        */
                v->dec_ready=0;au_reset(&v->au);
                v->stuck_pts_calls=0;v->last_decoded_pts_valid=0;
                goto push_done;
            }
            static int _frm=0;
            int _frames_this_call=0;
            if(_sp==0){
            while(avcodec_receive_frame(v->dec_ctx,v->frame)==0){
                _frames_this_call++;
                v->mpeg2_frames_since_flush++; /* DIAGNOSTIC -- see field doc */
                _frm++;if(_frm<=3)printf("[vidxc] frame%d decoded %dx%d\n",_frm,v->frame->width,v->frame->height);
                int w=v->frame->width,h=v->frame->height;
                AVRational src_sar=v->frame->sample_aspect_ratio;
                if(!v->enc_ctx){
                    if(!w||!h){av_frame_unref(v->frame);continue;}
                    if(!v->fps_confirmed){
                        /* Decoder is already running and correctly
                         * catching every IDR (see the fix that stopped
                         * gating VidXc creation on the fps probe) --
                         * only the ENCODER needs a real fps, and it's
                         * cheap to just wait here: drop this decoded
                         * frame and try again on the next one, until
                         * fps_probe_finalize_apply confirms the real
                         * rate (bounded by the probe's own window, not
                         * "wait for the next GOP" the way the old
                         * decoder-level gate was). This also means the
                         * encoder is never opened on a wrong/default
                         * fps in the first place, so no mismatch is
                         * ever possible and no teardown+reopen is
                         * needed.                                      */
                        av_frame_unref(v->frame);continue;
                    }
                    if(!vidxc_open_enc(v,w,h,src_sar,
                                       (enum AVColorSpace)v->frame->colorspace,
                                       (enum AVColorPrimaries)v->frame->color_primaries,
                                       (enum AVColorTransferCharacteristic)v->frame->color_trc)){
                        av_frame_unref(v->frame);continue;}
                }
                AVFrame*ef=av_frame_alloc();
                ef->format=AV_PIX_FMT_YUV420P;
                ef->width=w;ef->height=h;
                ef->sample_aspect_ratio=src_sar;
                ef->color_range=AVCOL_RANGE_MPEG; /* always — see full-
                    range conversion below and the doc on enc_ctx's
                    color_range assignment in vidxc_open_enc.          */
                ef->colorspace=v->frame->colorspace;
                ef->color_primaries=v->frame->color_primaries;
                ef->color_trc=v->frame->color_trc;
                av_frame_get_buffer(ef,0);
                /* Full-range -> limited-range fix (see VidXc.range_sws
                 * doc for the full root-cause explanation). Detect a
                 * full-range ("PC"/JPEG) source either via the
                 * yuvj4:2:0-family pixel formats (the classic signal —
                 * ffprobe shows this as "yuvj420p(pc,...)") or via an
                 * explicit AVCOL_RANGE_JPEG tag on frames already in a
                 * plain yuv420p format. Only THIS path needs an actual
                 * sws_scale rescale of pixel values; ordinary
                 * already-limited-range sources keep the cheap
                 * av_frame_copy() with zero extra cost or behavior
                 * change.                                              */
                enum AVPixelFormat sfmt=(enum AVPixelFormat)v->frame->format;
                int full_range=(sfmt==AV_PIX_FMT_YUVJ420P||sfmt==AV_PIX_FMT_YUVJ422P||
                                sfmt==AV_PIX_FMT_YUVJ444P||sfmt==AV_PIX_FMT_YUVJ440P||
                                v->frame->color_range==AVCOL_RANGE_JPEG);
                if(!full_range){
                av_frame_copy(ef,v->frame);
                }else{
                    if(!v->range_sws||v->range_sws_w!=w||v->range_sws_h!=h||
                       v->range_sws_srcfmt!=sfmt){
                        if(v->range_sws){sws_freeContext(v->range_sws);v->range_sws=NULL;}
                        v->range_sws=sws_getContext(w,h,sfmt,w,h,AV_PIX_FMT_YUV420P,
                                                     SWS_POINT,NULL,NULL,NULL);
                        v->range_sws_w=w;v->range_sws_h=h;v->range_sws_srcfmt=sfmt;
                        if(v->range_sws){
                            int brightness,contrast,saturation,srcRange,dstRange;
                            const int *inv_tbl,*fwd_tbl;
                            sws_getColorspaceDetails(v->range_sws,(int**)&inv_tbl,
                                &srcRange,(int**)&fwd_tbl,&dstRange,
                                &brightness,&contrast,&saturation);
                            const int*coeff=sws_getCoefficients(
                                v->frame->colorspace==AVCOL_SPC_BT709?
                                SWS_CS_ITU709:SWS_CS_ITU601);
                            /* srcRange=1 (full/PC) -> dstRange=0 (limited/TV):
                             * this is the actual value rescale (0-255 -> 16-235
                             * luma, 1-254 -> 16-240 chroma) that the old
                             * av_frame_copy() never did.                      */
                            sws_setColorspaceDetails(v->range_sws,coeff,1,
                                                      coeff,0,brightness,
                                                      contrast,saturation);
                        }
                    }
                    if(v->range_sws)
                        sws_scale(v->range_sws,(const uint8_t*const*)v->frame->data,
                                  v->frame->linesize,0,h,ef->data,ef->linesize);
                    else
                        av_frame_copy(ef,v->frame); /* sws alloc failed —
                            fall back to old (mistagged but non-crashing)
                            behavior rather than drop the frame.         */
                }
                /* Deinterlace: proper field-blend on all planes (Y, Cb, Cr).
                 * For interlaced source (1080i), blend odd lines with
                 * their neighbors to create progressive output.
                 * This eliminates combing artifacts on LG/Samsung.   */
                int interlaced=0;
#ifdef AV_FRAME_FLAG_INTERLACED
                interlaced=(v->frame->flags&AV_FRAME_FLAG_INTERLACED)!=0;
#else
                interlaced=v->frame->interlaced_frame;
#endif
                if(interlaced){
                    /* Process all 3 planes: Y (full res), Cb, Cr (half res) */
                    int plane_h[3]={h,h/2,h/2};
                    int plane_w[3]={w,w/2,w/2};
                    for(int pl=0;pl<3;pl++){
                        int ph=plane_h[pl],pw=plane_w[pl];
                        for(int y=1;y<ph-1;y+=2){
                            uint8_t*prev=ef->data[pl]+(y-1)*ef->linesize[pl];
                            uint8_t*cur =ef->data[pl]+y    *ef->linesize[pl];
                            uint8_t*next=ef->data[pl]+(y+1)*ef->linesize[pl];
                            for(int x=0;x<pw;x++)
                                cur[x]=(uint8_t)((prev[x]+next[x])>>1);
                        }
                    }
                }
                /* Use the DECODER's own output PTS for A/V sync, not the
                 * PTS of whatever AU was most recently fed in. This source
                 * has B-frames (decode order != presentation order), and
                 * libavcodec's H264 decoder already reorders frames into
                 * correct presentation order internally — frame->pts
                 * reflects that reordering correctly. v->au.pts is the
                 * PTS of the AU most recently PUSHED to the decoder
                 * (submission/decode order), which is NOT the same frame
                 * as the one just RECEIVED when B-frames are buffered
                 * inside the decoder's reorder queue. Using au.pts here
                 * silently stamped every output frame with a stale/wrong
                 * timestamp, producing a wildly out-of-order PTS sequence
                 * in the final TS (confirmed: 211522, 215122, 222322,
                 * 240322, 233122, 229522... instead of monotonically
                 * increasing) — exactly matching "plays like slow motion
                 * frame by frame in VLC" (VLC has to wait/seek across out
                 * of order timestamps), "macroblocks in LG" (a hardware
                 * decoder fed frames in the wrong presentation order),
                 * and "no playback in ExoPlayer" (strict players reject
                 * non-monotonic timestamps outright). Verified via a
                 * direct test harness using this exact AU-build/decode
                 * path: frame->pts came back perfectly monotonic
                 * (209722, 213322, 216922, 220522...) every single time. */
                int64_t src_pts=v->frame->pts!=AV_NOPTS_VALUE?v->frame->pts:AV_NOPTS_VALUE;
#if 0
                /* PTS wrap-normalize / sanity clamp — fixes a confirmed,
                 * exactly-reproduced bug: real production segments (169-
                 * 177, 6s apart) showed frame->pts periodically equal to
                 * the correct value plus EXACTLY 8589934592 (2^33 ticks,
                 * the standard 90kHz MPEG PTS wrap boundary ≈26.5h) — to
                 * sub-millisecond precision, three separate times, with
                 * every other segment correct. That magnitude of jump
                 * (from ~141s to ~95585s) makes the container's reported
                 * duration explode to ~1105s / ~26h for what's actually
                 * a normal 2s segment, which is exactly consistent with
                 * both reported symptoms: players either can't reconcile
                 * such an impossible timeline and drop/stutter the video,
                 * or reject the audio track outright as unplayable given
                 * how far its (correctly-timed, separately-confirmed)
                 * pts now sits from this exploded video timeline.
                 * au_push's raw PTS extraction has no wrap-add logic at
                 * all (verified — it's a plain 33-bit bitfield read), so
                 * this is originating inside libavcodec's own decode/
                 * reorder path, not this code's PES parsing. Rather than
                 * chase the exact internal libavcodec mechanism, this
                 * clamp is a direct, verifiable fix for the confirmed
                 * symptom: undo an apparent ±2^33 shift when doing so
                 * makes the delta from the last good frame sane again,
                 * and as a final backstop, refuse to pass through any
                 * frame-to-frame jump bigger than 10s of 90kHz ticks
                 * (900000) — no real GOP/segment boundary ever legitimately
                 * jumps a single decoded frame that far — holding at the
                 * last good pts plus one nominal frame duration instead.  */
                if(src_pts!=AV_NOPTS_VALUE){
                    if(v->last_good_pts_valid){
                        static const int64_t WRAP33=8589934592LL; /* 2^33 */
                        int64_t delta=src_pts-v->last_good_pts;
                        if(delta>(WRAP33/2))src_pts-=WRAP33;
                        else if(delta<-(WRAP33/2))src_pts+=WRAP33;
                        delta=src_pts-v->last_good_pts;
                        if(delta<0||delta>900000){
                            src_pts=v->last_good_pts+
                                (int64_t)90000*v->fpd/v->fpn;
                        }
                    }
                    v->last_good_pts=src_pts;v->last_good_pts_valid=1;
                }
#endif
                /* Stuck-PTS stall detection: DISABLED.
                 *
                 * This was added to catch a decoder stuck on stale/
                 * corrupted reference frames (real evidence: 3
                 * consecutive production segments with video PCR
                 * frozen at an identical value, alongside genuine
                 * H264 decode errors). Two attempts at this heuristic
                 * (first using <= for "non-increasing", then tightened
                 * to == for "exact repeat") were each followed by real,
                 * reported regressions — the second specifically
                 * causing VLC to freeze within 5 seconds, reliably,
                 * which could not be reproduced in testing. Since this
                 * logic writes to v->dec_ready (the same gate
                 * controlling whether ANY video frame passes through),
                 * a false positive here can fully block video output,
                 * not just degrade it — too severe a risk to keep
                 * active without being able to verify it's safe
                 * against real traffic in isolation first. The
                 * original bug this targeted is real but rare, and
                 * per the same real evidence that found it, appears to
                 * self-recover within a handful of segments even
                 * without this check — a missed catch costs a brief
                 * stutter; an unverified heuristic here has twice cost
                 * much worse. src_pts is still computed above (needed
                 * for the encoder's real PTS feed below) — only the
                 * stall-tracking logic itself is removed.             */
                /* Feed the encoder the REAL source pts (not an arbitrary
                 * counter) so that, even with real multithreading enabled
                 * (frame-level parallelism can buffer/delay frames
                 * internally — see ffmpeg's own threading docs), we can
                 * correctly recover which source frame a given output
                 * packet corresponds to by reading v->enc_pkt->pts back,
                 * rather than assuming output-call-order matches
                 * input-call-order (that assumption silently breaks under
                 * frame-threading and would reproduce the same class of
                 * out-of-order-timestamp bug already found and fixed on
                 * the decoder side earlier).                            */
                ef->pts=(src_pts!=AV_NOPTS_VALUE)?src_pts:(v->out_pts*90000LL*v->fpd/v->fpn);
                int this_frame_is_recovery=v->force_idr; /* capture before
                    it gets cleared below — this is the actual signal
                    for whether the resulting output packet's PCR
                    represents a real discontinuity (see vidxc_write
                    call further down, and its own doc, for why this
                    matters: VLC needs discontinuity_indicator set on
                    exactly this frame, ExoPlayer/LG tolerate it either
                    way).                                               */
                if(v->force_idr){
                    ef->pict_type=AV_PICTURE_TYPE_I;
#ifdef AV_FRAME_FLAG_KEY
                    ef->flags|=AV_FRAME_FLAG_KEY;
#else
                    ef->key_frame=1;
#endif
                    /* NOTE: deliberately NOT calling
                     * avcodec_flush_buffers(v->enc_ctx) here. That call
                     * is documented to interact badly with libx264's
                     * internal lookahead thread — confirmed via direct
                     * testing: it reproduces "lookahead thread is
                     * already stopped" warnings (483 in one 90s test
                     * run alone), recurring on every CC-recovery event.
                     * Forcing pict_type=I above already achieves what
                     * we actually need — x264 won't reference any
                     * prior (pre-gap) frame when encoding a true
                     * I-frame, regardless of internal buffer state —
                     * without touching the encoder's thread machinery.*/
                    v->force_idr=0;
                }
                if(avcodec_send_frame(v->enc_ctx,ef)==0){
                    uint8_t*combined=NULL;int clen=0;
                    int is_key=0;
                    int64_t out_pts=AV_NOPTS_VALUE;
                    while(avcodec_receive_packet(v->enc_ctx,v->enc_pkt)==0){
                        if(v->enc_pkt->flags&AV_PKT_FLAG_KEY)is_key=1;
                        out_pts=v->enc_pkt->pts; /* the pts THIS packet's
                                                     source frame actually
                                                     carried, correctly
                                                     tracked by libavcodec
                                                     across any internal
                                                     buffering delay      */
                        combined=realloc(combined,clen+v->enc_pkt->size);
                        memcpy(combined+clen,v->enc_pkt->data,v->enc_pkt->size);
                        clen+=v->enc_pkt->size;
                        av_packet_unref(v->enc_pkt);
                    }
                    if(combined&&clen>0){
                        src_pts=(out_pts!=AV_NOPTS_VALUE)?out_pts:src_pts;
                        /* Strip SEI NALs (type 6) — see comment above this
                         * block in vidxc_push for full rationale. libx264
                         * emits one of these (its own version string) on
                         * the first IDR; it triggers a known ExoPlayer/
                         * Media3 SeiReader/ReorderingBufferQueue crash on
                         * load. We don't need this NAL for anything, so
                         * just remove it from our own output entirely.   */
                        uint8_t*filtered=malloc(clen);
                        int flen=0;
                        int ci=0;
                        while(ci<clen){
                            int start_code_len=0;
                            if(ci+3<clen&&combined[ci]==0&&combined[ci+1]==0&&combined[ci+2]==1)
                                start_code_len=3;
                            else if(ci+4<clen&&combined[ci]==0&&combined[ci+1]==0&&
                                    combined[ci+2]==0&&combined[ci+3]==1)
                                start_code_len=4;
                            if(start_code_len==0){
                                /* not at a start code (shouldn't happen at
                                 * ci==0, but guard anyway) — copy single
                                 * byte and advance to stay safe           */
                                filtered[flen++]=combined[ci++];
                                continue;
                            }
                            int nal_start=ci;
                            int nal_hdr=ci+start_code_len;
                            int nal_type=(nal_hdr<clen)?(combined[nal_hdr]&0x1F):0;
                            /* find next start code (or end of buffer) to
                             * know where this NAL ends                   */
                            int next=nal_hdr;
                            while(next+2<clen){
                                if(combined[next]==0&&combined[next+1]==0&&
                                   (combined[next+2]==1||
                                    (next+3<clen&&combined[next+2]==0&&combined[next+3]==1)))
                                    break;
                                next++;
                            }
                            int nal_end=(next+2<clen)?next:clen;
                            if(nal_type!=6){
                                memcpy(filtered+flen,combined+nal_start,nal_end-nal_start);
                                flen+=nal_end-nal_start;
                            }
                            ci=nal_end;
                        }
                        if(is_key)v->got_keyframe=1;
                        int64_t vpts=src_pts!=AV_NOPTS_VALUE?src_pts:v->out_pts*90000LL*v->fpd/v->fpn;
                        /* Write every frame the encoder produces — including
                         * the very first one. This flag used to gate on
                         * v->enc_started, but that flag is only set by the
                         * CALLER (pkt_process), and only AFTER this function
                         * (vidxc_push) returns. On the very first IDR, that
                         * means enc_started is still 0 at this exact point
                         * — so the first frame (the ONLY one carrying the
                         * encoder's SPS/PPS, since repeat_headers=1 only
                         * repeats them on IDRs, and this IS the first IDR)
                         * was silently dropped here and never written to
                         * v->out at all. Every later IDR worked fine
                         * because by then enc_started was already 1 from
                         * a previous call. Confirmed via real broadcast
                         * capture: segment 0 (the very first segment) had
                         * zero SPS/PPS NALs in its entire video payload —
                         * 0 of 49 frames decoded — while segments 1-3
                         * decoded perfectly. This exactly matches "video
                         * freezes/macroblocks from the start, ExoPlayer
                         * refuses playback" — a player loading segment 0
                         * first hits a stream with no parameter sets at
                         * all. is_key/got_keyframe is still used by the
                         * caller to decide whether THIS frame should
                         * trigger seg_open()/segment-cut; that logic is
                         * unaffected by removing the write-side gate here. */
                        vidxc_write(v,filtered,flen,pid,vpts,vpts,this_frame_is_recovery);
                        free(filtered);
                        free(combined);
                    }
                }
                v->out_pts++;
                av_frame_free(&ef);av_frame_unref(v->frame);
            }
            if(_frames_this_call>0){
                v->stall_calls=0;
            } else {
                /* send_packet succeeded but produced no frame this
                 * call. This is NORMAL and expected on sources using
                 * B-frames (confirmed this source does: direct check
                 * found B/I/P all present) — the decoder legitimately
                 * buffers a few frames internally for reorder before
                 * its first output. Only treat this as a genuine stall
                 * (and reset dec_ready to force I-slice re-validation)
                 * after a SUSTAINED run with zero output — long enough
                 * to rule out normal reorder delay (bounded to a
                 * handful of frames, never dozens), but short enough
                 * to recover quickly from a real post-CC-drop decode
                 * stall (the actual bug this is fixing — see the
                 * send_packet-error branch above for full rationale,
                 * confirmed against two real production logs where
                 * this exact failure mode froze segment output for
                 * 30+ minutes with no recovery).                      */
                v->stall_calls++;
                if(v->stall_calls>=30){
                    v->dec_ready=0;v->stall_calls=0;
                    v->stuck_pts_calls=0;v->last_decoded_pts_valid=0;
                }
            }
            /* Real, sustained reference-frame corruption check — runs
             * UNCONDITIONALLY regardless of whether frames were
             * produced this call, since the confirmed real bug shows
             * frames DO keep being produced while corrupted (that is
             * exactly why the zero-frame stall check above never
             * caught this in the first place). Directly observed via
             * libavcodec's own log output (corrupt_errors_since_check,
             * fed by corruption_log_callback) — not an inferred
             * heuristic like the disabled stuck-PTS check. Confirmed
             * against real production evidence: a single CC drop left
             * "mmco: unref short failure"/"reference picture missing"
             * errors escalating across 6+ consecutive segments with
             * zero further CC drops to ever trigger recovery any other
             * way. Threshold of 5 consecutive calls (much tighter than
             * stall_calls' 30) is appropriate here because this is a
             * direct signal — a real corruption error logged
             * repeatedly is unambiguous, unlike "zero frames" which
             * has a legitimate explanation in normal B-frame reorder
             * delay.                                                  */
            if(v->corrupt_errors_since_check>0){
                v->corrupt_calls++;
                if(v->corrupt_calls>=5){
                    v->dec_ready=0;v->corrupt_calls=0;
                    v->stall_calls=0;v->stuck_pts_calls=0;
                    v->last_decoded_pts_valid=0;
                }
            } else {
                v->corrupt_calls=0;
            }
            v->corrupt_errors_since_check=0;
            }}
        au_reset(&v->au);
    }
push_done:
    au_push(&v->au,ts_pkt,is_au_boundary);
    return v->out_len/TS_SZ;
}

/* ════════════════════════════════════════════════════════════════
   AES-128-CBC
   ════════════════════════════════════════════════════════════════ */
typedef struct{AES_KEY enc;uint8_t iv0[AES_BLK],iv[AES_BLK];
               uint8_t buf[WRITE_BATCH*TS_SZ+AES_BLK*2];int n;}Aes;
static void aes_init(Aes*a,const uint8_t*k,const uint8_t*iv){
    AES_set_encrypt_key(k,128,&a->enc);
    memcpy(a->iv0,iv,AES_BLK);memcpy(a->iv,iv,AES_BLK);a->n=0;}
static void aes_reset(Aes*a){memcpy(a->iv,a->iv0,AES_BLK);a->n=0;}
static void aes_flush(Aes*a,int fd,int fin){
    int n=a->n;
    if(fin){int p=AES_BLK-(n%AES_BLK);if(!p)p=AES_BLK;memset(a->buf+n,p,p);n+=p;}
    else n=(n/AES_BLK)*AES_BLK;
    if(n<=0)return;
    uint8_t out[sizeof(a->buf)];
    AES_cbc_encrypt(a->buf,out,n,&a->enc,a->iv,AES_ENCRYPT);
    ssize_t r=write(fd,out,n);(void)r;
    int rem=fin?0:a->n-n;
    if(rem>0)memmove(a->buf,a->buf+n,rem);a->n=rem;}
static void aes_push(Aes*a,int fd,const uint8_t*p){
    memcpy(a->buf+a->n,p,TS_SZ);a->n+=TS_SZ;
    if(a->n>=WRITE_BATCH*TS_SZ)aes_flush(a,fd,0);}

/* ════════════════════════════════════════════════════════════════
   CHANNEL OPTIONS & STATE
   ════════════════════════════════════════════════════════════════ */
typedef struct{
    int   audio_transcode,video_transcode;
    int   fix_mp2,fix_interlace,pts_reset;
    float video_crf;
    int   audio_bitrate;   /* AAC target bitrate, default 128000        */
    int   audio_map;       /* 1-indexed position of a SINGLE audio track
                               to select (by PMT order); 0 = not set,
                               meaning select ALL discovered audio
                               tracks instead (default multi-track
                               behavior). When set, every OTHER audio
                               track is dropped entirely from output.   */
    char  audio_coder[16]; /* AAC encoder algorithm: "twoloop" (default,
                               best quality/bit) or "fast" (measured
                               ~2.4x lower CPU, slightly less optimal
                               bit allocation at the same bitrate — see
                               audxc_init for benchmark numbers). Native
                               ffmpeg AAC encoder only; libfdk_aac is not
                               available in this build environment and
                               aac_mode=vbrN is an libfdk_aac-specific
                               option name that the native encoder does
                               not implement (native AAC is CBR-only,
                               controlled via bit_rate).                */
    int   simple_cut;      /* 0 (default) = has_keyframe: Exp-Golomb
                               slice_type check, correctly recognizes
                               non-standard nal_type=1 functional
                               keyframes on sources that need it. 1 =
                               scan_nal-only: raw NAL type check only
                               (matches the older v23 codebase's
                               cut logic), no Exp-Golomb parsing. Only
                               affects the simple-copy/audio_transcode
                               segment-cut decision — video_transcode's
                               internal decoder gate (vidxc_push) always
                               uses the full has_keyframe/is_islice
                               check regardless of this setting, since
                               it has a stricter requirement (decoder
                               needs a REAL safe-to-decode point, not
                               just a reasonable place to cut a file)
                               and switching that gate risks the decoder
                               never starting at all on sources that
                               only use nal_type=1, not real IDRs.      */
    int   idr_rewrite;     /* 0 (default) = off, leave the cut-point
                               slice NAL exactly as the source sent it.
                               1 = rewrite the cut-point slice's NAL
                               type from 1 to 5 (real IDR) in the
                               simple-copy/audio_transcode segment-cut
                               path only, on the EXACT packet identified
                               as a cut point (never elsewhere in the
                               stream). See rewrite_slice_to_idr() for
                               full rationale: this source's encoder
                               never emits real type-5 IDRs (confirmed:
                               zero across 8 real consecutive captured
                               segments), only Exp-Golomb-confirmed
                               all-intra type-1 slices, which leaves
                               every segment boundary structurally
                               unable to tell a decoder to discard
                               prior reference-buffer state — directly
                               confirmed as the cause of "mmco: unref
                               short failure"/"reference picture
                               missing" errors on every one of those 8
                               segments. Off by default until verified
                               against a real player on this source,
                               since the rewrite has one acknowledged,
                               untested gap (frame_num/POC continuity
                               at the rewritten boundary — see the
                               function doc for detail).                */
}ChOpts;

/* ----------------------------------------------------------------
   v23-COMPATIBLE SIMPLE CUT (simple_cut=1)
   ---------------------------------------------------------------- */
/* Implemented inline at each call site via nal_type(p)==5||==20 —
 * matches v23's scan_nal()'s IDR detection (raw NAL type only, no
 * Exp-Golomb slice_type parsing). See simple_cut option doc above
 * for full rationale.                                               */

/* Multi-audio-track support: each selected/transcoded source audio
 * track gets its own slot. src_pid is the PID as it appears in the
 * SOURCE PMT; out_pid is the PID we actually write transcoded or
 * passed-through output on. For the first/only track in the common
 * single-track case, out_pid==src_pid (reusing the original PID for
 * a 1:1 replace/passthrough is simplest and most compatible) — only
 * additional tracks beyond the first need a distinct, synthesized
 * output PID, since two different streams can't share one PID.      */
typedef struct{
    uint16_t src_pid,out_pid;
    uint8_t  stream_type;
    AudXc    audxc;
    uint8_t  cc;
    uint64_t write_loop_entries; /* number of times the audio dispatch
        code's pkt_write loop body actually executed (i.e. nout>0 was
        true and at least one TS-packet-sized chunk was written) -
        for the SIGUSR1 dump, to pin down precisely where real,
        confirmed-produced AAC output (see audxc.push_calls_with_
        output) might be getting lost before it reaches the segment
        file. Confirmed real gap: a production dump showed
        push_calls_with_output climbing (real output being produced)
        at the exact same moment 7 consecutive real segments showed
        zero audio packets — this counter narrows down whether the
        write loop itself is even running.                            */
    uint64_t write_loop_packets; /* total TS-packet-sized chunks
        actually passed to pkt_write across all write_loop_entries -
        lets the dump distinguish "loop ran once but wrote 0 packets"
        from "loop ran and wrote many packets" (which would mean the
        bug is downstream of pkt_write itself, e.g. seg_fd handling).*/
}AudSlot;

typedef struct{
    /* config */
    char    mcast[512],dir[256],name[64],iface[64]; /* mcast widened from
        64->512 to also hold a full HLS input URL (http://host:port/path)
        when c->hls_input is set -- existing UDP multicast addresses are
        far shorter and completely unaffected by this.                  */
    char    keyfile[256],keyuri[512],iv_hex[33];
    int     port; uint64_t start_seq; ChOpts opts;
    /* HLS INPUT support: when c->mcast holds an "http://" URL instead of
     * a multicast address (detected automatically in main()/conf_load,
     * no separate CLI flag needed), this channel is served by
     * hls_input_thread() instead of the normal UDP nic_thread path. All
     * downstream processing (pkt_process, segment cutting, audio/video
     * transcode, OUTPUT encryption via c->keyfile/keyuri/iv_hex) is
     * completely unchanged -- HLS input just becomes a second way to
     * get TS packets fed into the exact same pipeline. If the SOURCE
     * playlist itself is encrypted (#EXT-X-KEY present), this decrypts
     * it independently using ITS OWN key/IV as declared in that
     * playlist -- unrelated to and never conflated with this channel's
     * own OUTPUT encryption settings.                                   */
    int      hls_input;
    uint64_t hls_last_seq; int hls_last_seq_ok; /* highest source media-
        sequence number already processed -- only strictly-greater
        sequence numbers get fetched on each subsequent poll, so a
        segment is never processed twice even if it's still listed on
        the next playlist refresh.                                      */
    char     hls_key_uri[512]; uint8_t hls_key[16]; int hls_key_ok;
        /* cached source decryption key -- re-fetched only when the
         * playlist's #EXT-X-KEY URI actually changes, not on every
         * segment/poll.                                                 */
    /* socket */
    int     fd;
    /* PSI — always write patched cached copy, never raw source */
    uint8_t pat[TS_SZ],pmt[TS_SZ];
    int     pat_ok,pmt_ok;
    uint16_t pmt_pid,vid_pid,pcr_pid;
    /* fps auto-detection for video_transcode -- replaces the old
     * hardcoded 25/1 passed to vidxc_init. Measures the real inter-
     * frame period directly from raw video-PID PES pts deltas (needs
     * no VUI/SPS parsing, which broadcast streams often omit or encode
     * inconsistently) and derives fpn/fpd from the MEDIAN of 16
     * samples once collected -- median rather than mean specifically
     * to stay robust against an occasional glitched/duplicated pts
     * without a separate outlier-rejection pass. video_transcode init
     * is deferred until this probe completes, so the encoder always
     * opens with a real measured fps, never a guess.                  */
    int64_t fps_probe_last_pts; int fps_probe_have_last;
    int64_t fps_probe_deltas[16]; int fps_probe_n;
    int     fps_detected_fpn, fps_detected_fpd, fps_probe_done;
    int     fps_probe_mbs_only; /* -1=unknown yet, 0=interlaced,
        1=progressive -- read from the SPS's own frame_mbs_only_flag
        (see sps_read_frame_mbs_only) while the probe is collecting
        samples, so a genuinely-interlaced source's measured FIELD rate
        can be corrected down to the real FRAME rate before use.       */
    int     fps_probe_ambiguous; /* set once the 16-sample timing probe
        has finished and landed on a rate (50/1, 60/1, 60000/1001) that
        is indistinguishable, by timing alone, from the corresponding
        halved interlaced rate -- triggers the extended SPS search
        below instead of finalizing immediately on a maybe-wrong guess.
        Confirmed root cause of a real misdetection: a 25fps interlaced
        (top-field-first) 720x576 source measured as a clean 50/1 by
        the 16-sample timing probe, but no SPS happened to fall inside
        that narrow 16-packet window (SPS is only sent once per
        GOP/IDR, far less often than every packet), so mbs_only stayed
        -1 and the halving correction below never fired, locking the
        channel into the wrong fps for its whole run.                  */
    int     fps_probe_extra_pkts; /* counts packets scanned during that
        extended search, capped so a source that genuinely never repeats
        an SPS within a bounded window doesn't stall fps detection
        forever -- falls through to the safety-net halving below once
        the cap is hit with mbs_only still unresolved.                  */
    int     fps_probe_pending_fpn, fps_probe_pending_fpd;
    int64_t fps_probe_pending_rep; /* the timing-derived candidate,
        held while fps_probe_ambiguous is set and finalizing is
        deferred -- consumed by fps_probe_try_finalize_deferred.        */
    uint8_t  vid_stream_type;
    /* Multi-audio-track support: see AudSlot definition above Ch for
     * full rationale. n_aud_active is how many slots are in use.    */
    AudSlot  aud[MAX_AUD_TRACKS];
    int      n_aud_active;
    /* Audio PIDs discovered in the PMT but explicitly excluded by
     * audio_map (the unselected tracks that must be dropped from
     * output entirely, not merely omitted from PMT metadata).        */
    uint16_t aud_dropped[MAX_AUD_TRACKS];
    int      n_aud_dropped;
    /* SPS+PPS cache injected at every seg_open */
    uint8_t  spspps[8][TS_SZ];int spspps_n;
    /* segment */
    int     seg_fd; uint64_t seg_seq;
    double  dur[DUR_SLOTS]; uint64_t seg_size[DUR_SLOTS];
    int     target_dur,target_dur_fixed;
    /* AES */
    int     aes_on; Aes aes;
    /* write batch */
    uint8_t wb[WRITE_BATCH*TS_SZ]; int wb_n;
    /* timing */
    struct timespec seg_start; int seg_timing_ok;
    int64_t         last_pcr27;    /* last 27MHz PCR seen on pcr_pid,
        for interpolating PCR values into gap-stuffing packets.
        -1 = not yet seen (use plain null packets instead).           */
    int             cc_drop_skip;  /* 1 = discard video PID packets until
        next PUSI after a CC gap, to avoid feeding a corrupted mid-PES
        continuation to the decoder. Set on CC drop in simple-copy mode,
        cleared on the next vid_pid packet with PUSI=1.               */
    /* Last time CC-drop recovery was actually triggered (decoder
     * flush + dec_ready reset) — used for a short cooldown so a burst
     * of many CC drops in the same fraction of a second doesn't keep
     * cancelling an in-progress recovery before the decoder gets a
     * real chance to produce output and let a segment open. See the
     * cooldown check at the CC-drop handler for full rationale.       */
    struct timespec last_recovery_trigger; int last_recovery_trigger_valid;
    /* state */
    int     started,recovering;
    /* Hybrid IDR->PUSI fallback timers — see MAX_START_WAIT_SECS doc.
     * *_wait_ok=1 means *_wait_start holds the timestamp of the FIRST
     * candidate video PUSI seen since we started waiting (for startup:
     * since !c->started became true for this connection; for recovery:
     * since c->recovering became 1) — reset to 0 whenever a wait period
     * ends, whether by a qualifying keyframe arriving or by the
     * fallback itself firing, and whenever started/recovering get
     * reset elsewhere (full channel reset / reconnect).                */
    struct timespec start_wait_start; int start_wait_ok;
    struct timespec recover_wait_start; int recover_wait_ok;
    /* v23-compatible permanent open-GOP fallback — see
     * OPEN_GOP_MISS_THRESHOLD doc. Once open_gop=1, the segment-cut
     * logic stops trying to detect IDR/has_keyframe() cut points
     * entirely and cuts on any video PUSI once normal segment timing
     * allows, exactly like v23's idr_off. seg_wait_misses counts
     * consecutive MAX_SEG_WAIT_SECS forced cuts (see the cut block)
     * and is reset to 0 the moment a real is_cut_point cut succeeds,
     * so a channel that regains real keyframes stays out of open-GOP
     * unless it starts missing again.                                 */
    int     open_gop,seg_wait_misses;
    /* CC */
    uint8_t cc_last[8192]; uint32_t cc_errors;
    /* PTS reset */
    int     pts_base_ok; int64_t pts_base;
    /* transcoders */
    AudXc   audxc; uint8_t aud_cc;
    AudScratch aud_scratch; /* shared scratch buffers for ALL of this
        channel's active audio tracks (see AudScratch doc for why this
        is safe to share: one channel's audxc_push calls across its
        tracks are always sequential, never concurrent). Lives here
        (once per channel) instead of inside AudXc (which would be
        once per track, up to MAX_AUD_TRACKS=8x more).                 */
    VidXc  *vidxc;  /* heap-allocated lazily, only for channels that
        set video_transcode=1 -- embedded by value (VidXc vidxc;) makes
        EVERY one of the MAX_CHANNELS=512 static g_ch[] slots cost
        ~1.3MB whether or not that channel uses video_transcode at all
        (due to VidXc's 1MB AuBuf, see AU_MAX): ~648MB committed at
        process startup unconditionally. As a pointer, channels that
        never set video_transcode=1 (the common case at high channel
        counts) cost only 8 bytes here instead of ~1MB; the real
        allocation only happens for channels that actually need it, in
        pkt_process's video_transcode init block. Every call site below
        is c->vidxc->field (NULL-guarded), not c->vidxc.field -- this
        has been converted back to a pointer at least once already
        after a copy-paste from an older snapshot reintroduced the
        by-value struct as a regression; if c->vidxc. (dot, not arrow)
        shows up anywhere again, that's the same regression back.    */
    /* stats */
    uint64_t dgrams,pkts,segs;
    struct timespec last_pkt;
    uint64_t snap_pkts,snap_bytes,snap_cc;
    struct timespec snap_time;
    EvtRing  evring;
}Ch;

static volatile sig_atomic_t g_stop=0;
static volatile sig_atomic_t g_dump_requested=0; /* set by SIGUSR1 handler;
    consumed by the stats thread, which dumps full per-channel audio/video
    state to help diagnose long-run issues (e.g. "audio silently stops
    after days, video stays fine, restart fixes it") that are impractical
    to reproduce on demand — trigger with `kill -USR1 <pid>` right before
    restarting, so the dump captures the actual failing state instead of
    a fresh post-restart one.                                            */
static Ch g_ch[MAX_CHANNELS]; static int g_nch=0;

/* corruption_log_callback — FIXED version (was declared earlier in the
   file using a racy shared global g_corruption_target; see the comment
   left at that old location for the full bug writeup). Matches each
   libavcodec error line directly to the channel whose dec_ctx produced
   it, via the avcl pointer libavcodec already passes us — no shared
   mutable state, so this is correct even with many video_transcode
   channels decoding concurrently across multiple NIC threads.         */
static void corruption_log_callback(void*avcl,int level,const char*fmt,va_list vl){
    if(level>AV_LOG_ERROR)return; /* only care about real errors, not
        info/debug/warning-level chatter - keeps this cheap on the
        overwhelming majority of log calls during normal operation    */
    char buf[256];
    int n=vsnprintf(buf,sizeof buf,fmt,vl);
    if(n<=0)return;
    if(strstr(buf,"mmco: unref short failure")||
       strstr(buf,"reference picture missing")){
        for(int i=0;i<g_nch;i++){
            Ch*c=&g_ch[i];
            if(c->opts.video_transcode && c->vidxc &&
               c->vidxc->dec_ctx==(AVCodecContext*)avcl){
                c->vidxc->corrupt_errors_since_check++;
                break;
            }
        }
    }
}
static int g_del=KEEP_EXTRA;
static void onsig(int s){(void)s;g_stop=1;}
static void onsig_dump(int s){(void)s;g_dump_requested=1;}

/* Returns the index of the active audio slot whose out_pid (==src_pid,
 * per the design — see AudSlot) matches pid, or -1 if pid isn't one of
 * this channel's currently selected audio tracks. Used everywhere the
 * old code did a single `pid==c->aud_pid` check.                      */
static inline int aud_slot_for_pid(Ch*c,uint16_t pid){
    for(int i=0;i<c->n_aud_active;i++)
        if(c->aud[i].src_pid==pid)return i;
    return -1;}

/* True if pid is an audio track that was discovered in the PMT but
 * explicitly excluded by audio_map — its packets must be actively
 * dropped from output, not merely omitted from PMT metadata.         */
static inline int aud_is_dropped(Ch*c,uint16_t pid){
    for(int i=0;i<c->n_aud_dropped;i++)
        if(c->aud_dropped[i]==pid)return 1;
    return 0;}

/* ════════════════════════════════════════════════════════════════
   TS HELPERS
   ════════════════════════════════════════════════════════════════ */
static inline uint16_t ts_pid(const uint8_t*p){return((p[1]&0x1F)<<8)|p[2];}
static inline int ts_pusi(const uint8_t*p){return(p[1]>>6)&1;}
static inline int ts_afc(const uint8_t*p){return(p[3]>>4)&3;}
static inline int ts_poff(const uint8_t*p){
    int a=ts_afc(p),o;
    if(a==1)o=4;else if(a==3)o=5+(int)p[4];else return -1;
    return o<TS_SZ?o:-1;}

static uint16_t pat_parse(const uint8_t*p){
    int o=ts_poff(p);if(o<0||o>=TS_SZ)return 0;
    int b=o+1+p[o];if(b+8>=TS_SZ||p[b]!=0)return 0;
    int e=b+3+(((p[b+1]&0xF)<<8)|p[b+2])-4;
    for(int q=b+8;q+3<=e&&q+3<TS_SZ;q+=4){
        uint16_t pn=(p[q]<<8)|p[q+1];
        uint16_t pp=((p[q+2]&0x1F)<<8)|p[q+3];
        if(pn)return pp;}
    return 0;}
static int is_video(uint8_t t){return t==0x01||t==0x02||t==0x10||t==0x1B||t==0x24||t==0x27;}
static int is_audio(uint8_t t){return t==0x03||t==0x04||t==0x06||t==0x0F||t==0x11||t==0x81||t==0x87;}
/* Stream types our audio_transcode pipeline can actually decode:
 * 0x03/0x04 (MPEG-1/2 audio, MP2/MP3 via mpg123), 0x0F (already AAC, no
 * decode needed), and 0x06/0x87/0x81 (AC3/E-AC3, decoded via libavcodec
 * and downmixed to stereo if the source is 5.1/surround — see
 * audxc_push_ac3). NOTE on 0x81: this is a private-data tag that
 * different muxers use for different codecs (some use it for DTS, not
 * AC3) — treating it as AC3 here is confirmed correct for this
 * deployment's specific encoders, not a universally safe assumption.
 * HE-AAC LATM (0x11) remains unsupported: recognized as "audio" for
 * PMT parsing purposes but picking it when a decodable alternative
 * exists would silently feed unsupported bytes into a decoder that
 * can't handle them.                                                  */
static int is_decodable_audio(uint8_t t){return t==0x03||t==0x04||t==0x0F||t==0x06||t==0x87||t==0x81;}
static int pmt_parse(const uint8_t*p,uint16_t*pcr,uint16_t*vid,uint16_t*aud,uint8_t*vst,uint8_t*aud_st){
    int o=ts_poff(p);if(o<0||o+1>=TS_SZ)return 0;
    int b=o+1+p[o];if(b+12>=TS_SZ||p[b]!=0x02)return 0;
    *pcr=((p[b+8]&0x1F)<<8)|p[b+9];
    int pil=((p[b+10]&0xF)<<8)|p[b+11];
    int sec=((p[b+1]&0xF)<<8)|p[b+2];
    int es=b+12+pil,ee=b+3+sec-4;*vid=*aud=0;if(vst)*vst=0;if(aud_st)*aud_st=0;
    uint16_t fallback_aud=0;uint8_t fallback_st=0;
    while(es+4<TS_SZ&&es+4<=ee){
        uint8_t st=p[es];uint16_t sp=((p[es+1]&0x1F)<<8)|p[es+2];
        int el=((p[es+3]&0xF)<<8)|p[es+4];
        if(!*vid&&is_video(st)){*vid=sp;if(vst)*vst=st;}
        /* Prefer a decodable audio stream; remember the first audio
         * stream of ANY type as a fallback only in case nothing
         * decodable is found anywhere in the PMT.                    */
        if(!*aud&&is_decodable_audio(st)){*aud=sp;if(aud_st)*aud_st=st;}
        else if(!fallback_aud&&is_audio(st)){fallback_aud=sp;fallback_st=st;}
        es+=5+el;}
    if(!*aud&&fallback_aud){*aud=fallback_aud;if(aud_st)*aud_st=fallback_st;}
    return *pcr>0;}

typedef struct{
    uint16_t pid;
    uint8_t  stream_type;
}AudTrack;
/* Discover ALL audio elementary streams in the PMT, in the order they
 * appear (this order is what audio_map=N indexes into, 1-based, per
 * the position-based selection design). Returns the number found
 * (capped at MAX_AUD_TRACKS — real broadcasts essentially never have
 * more than a handful of audio tracks).                              */
static int pmt_parse_all_audio(const uint8_t*p,AudTrack*tracks){
    int o=ts_poff(p);if(o<0||o+1>=TS_SZ)return 0;
    int b=o+1+p[o];if(b+12>=TS_SZ||p[b]!=0x02)return 0;
    int pil=((p[b+10]&0xF)<<8)|p[b+11];
    int sec=((p[b+1]&0xF)<<8)|p[b+2];
    int es=b+12+pil,ee=b+3+sec-4;int n=0;
    while(es+4<TS_SZ&&es+4<=ee&&n<MAX_AUD_TRACKS){
        uint8_t st=p[es];uint16_t sp=((p[es+1]&0x1F)<<8)|p[es+2];
        int el=((p[es+3]&0xF)<<8)|p[es+4];
        if(is_audio(st)){tracks[n].pid=sp;tracks[n].stream_type=st;n++;}
        es+=5+el;}
    return n;}

static uint32_t crc32_mpeg(const uint8_t*d,int len){
    uint32_t crc=0xFFFFFFFF;
    for(int i=0;i<len;i++){
        crc^=(uint32_t)d[i]<<24;
        for(int b=0;b<8;b++)crc=(crc&0x80000000)?(crc<<1)^0x04C11DB7:(crc<<1);}
    return crc;}

/* Patch PMT audio stream_type to 0x0F (AAC), and strip any descriptors
 * attached to that elementary stream entry (e.g. a registration_
 * descriptor explicitly naming "AC-3" — common on real AC3 sources,
 * see rationale above this function). Leaving such a descriptor in
 * place after changing stream_type causes demuxers that trust
 * descriptors over the raw stream_type byte to keep misidentifying
 * the (now-AAC) stream by its original codec.
 * Handles 0x03 (MPEG-1 Audio), 0x04 (MPEG-2 Audio/MP2), and now
 * 0x06/0x87/0x81 (AC3/E-AC3, transcoded via the libavcodec-based AC3
 * decode path — see audxc_push_ac3).
 * Recalculates CRC32.                                          */
static int pmt_patch_audio(uint8_t*pkt,uint16_t aud_pid){
    int o=ts_poff(pkt);if(o<0||o+1>=TS_SZ)return 0;
    int b=o+1+pkt[o];if(b+12>=TS_SZ||pkt[b]!=0x02)return 0;
    int sec=((pkt[b+1]&0xF)<<8)|pkt[b+2];
    int pil=((pkt[b+10]&0xF)<<8)|pkt[b+11];
    int es=b+12+pil,ee=b+3+sec-4;int patched=0;int removed_bytes=0;
    while(es+4<TS_SZ&&es+4<=ee){
        uint8_t st=pkt[es];
        uint16_t sp=((pkt[es+1]&0x1F)<<8)|pkt[es+2];
        int el=((pkt[es+3]&0xF)<<8)|pkt[es+4];
        if(sp==aud_pid&&(st==0x03||st==0x04||st==0x06||st==0x87||st==0x81)){
            static int logged=0;
            if(!logged){
                printf("[pmt] audio pid=0x%04X type=0x%02X -> 0x0F (AAC)"
                       "%s\n",sp,st,el>0?" (descriptors stripped)":"");
                logged=1;}
            pkt[es]=0x0F;patched=1;
            if(el>0){
                /* Strip descriptors: shift everything after this
                 * entry's descriptor bytes left by el, closing the
                 * gap, and set this entry's ES_info_length to 0.      */
                int desc_start=es+5;
                int tail_start=desc_start+el;
                int tail_len=TS_SZ-tail_start;
                memmove(pkt+desc_start,pkt+tail_start,tail_len);
                memset(pkt+TS_SZ-el,0xFF,el);
                pkt[es+3]&=0xF0; /* ES_info_length high nibble -> 0 */
                pkt[es+4]=0x00;  /* ES_info_length low byte -> 0     */
                removed_bytes+=el;
                ee-=el;
                el=0; /* this entry's descriptors are gone now       */
            }}
        es+=5+el;}
    if(!patched)return 0;
    int new_sec=sec-removed_bytes;
    pkt[b+1]=(uint8_t)((pkt[b+1]&0xF0)|((new_sec>>8)&0xF));
    pkt[b+2]=(uint8_t)(new_sec&0xFF);
    int coff=b+3+new_sec-4;if(coff+4>TS_SZ)return 1;
    uint32_t crc=crc32_mpeg(pkt+b,new_sec-1);
    pkt[coff+0]=(crc>>24)&0xFF;pkt[coff+1]=(crc>>16)&0xFF;
    pkt[coff+2]=(crc>>8)&0xFF; pkt[coff+3]=(crc)&0xFF;
    return 1;}

/* Patch PMT video stream_type to 0x1B (H264) when video_transcode=1 */
/* Remove every audio elementary stream entry from the PMT except the
 * one whose PID equals keep_pid (use keep_pid=0 to remove ALL audio
 * entries, e.g. if audio_map pointed at a nonexistent track). Shifts
 * remaining bytes to close the gap and recomputes section_length+CRC.
 * Only ever removes bytes, so the result always fits in one 188-byte
 * TS packet — no multi-packet PMT handling is required.               */
static int pmt_strip_unselected_audio(uint8_t*pkt,uint16_t keep_pid){
    int o=ts_poff(pkt);if(o<0||o+1>=TS_SZ)return 0;
    int b=o+1+pkt[o];if(b+12>=TS_SZ||pkt[b]!=0x02)return 0;
    int sec=((pkt[b+1]&0xF)<<8)|pkt[b+2];
    int pil=((pkt[b+10]&0xF)<<8)|pkt[b+11];
    int es=b+12+pil,ee=b+3+sec-4;int removed_bytes=0;
    while(es+4<TS_SZ&&es+4<=ee){
        uint8_t st=pkt[es];
        uint16_t sp=((pkt[es+1]&0x1F)<<8)|pkt[es+2];
        int el=((pkt[es+3]&0xF)<<8)|pkt[es+4];
        int entry_len=5+el;
        if(is_audio(st)&&sp!=keep_pid){
            /* Remove this entry: shift everything after it left by
             * entry_len bytes, shrinking the section in place.        */
            int tail_start=es+entry_len;
            int tail_len=TS_SZ-tail_start; /* conservative — copy to end
                of packet; bytes past the real section end are stuffing
                /already-irrelevant and get naturally truncated by the
                shrunk section_length written below anyway.            */
            memmove(pkt+es,pkt+tail_start,tail_len);
            /* Zero the now-unused tail so no stale entry bytes linger
             * past the new (shorter) section in the packet buffer.    */
            memset(pkt+TS_SZ-entry_len,0xFF,entry_len);
            removed_bytes+=entry_len;
            ee-=entry_len;
            /* Don't advance es — the next entry has shifted into this
             * same position and must also be checked.                */
            continue;
        }
        es+=entry_len;}
    if(!removed_bytes)return 0;
    int new_sec=sec-removed_bytes;
    pkt[b+1]=(uint8_t)(((pkt[b+1]&0xF0))|((new_sec>>8)&0xF));
    pkt[b+2]=(uint8_t)(new_sec&0xFF);
    int coff=b+3+new_sec-4;if(coff+4>TS_SZ)return 1;
    uint32_t crc=crc32_mpeg(pkt+b,new_sec-1);
    pkt[coff+0]=(crc>>24)&0xFF;pkt[coff+1]=(crc>>16)&0xFF;
    pkt[coff+2]=(crc>>8)&0xFF; pkt[coff+3]=(crc)&0xFF;
    return 1;}


/* Patch PMT video stream_type to 0x1B (H264) when video_transcode=1 */
static int pmt_patch_video(uint8_t*pkt,uint16_t vid_pid){
    int o=ts_poff(pkt);if(o<0||o+1>=TS_SZ)return 0;
    int b=o+1+pkt[o];if(b+12>=TS_SZ||pkt[b]!=0x02)return 0;
    int sec=((pkt[b+1]&0xF)<<8)|pkt[b+2];
    int pil=((pkt[b+10]&0xF)<<8)|pkt[b+11];
    int es=b+12+pil,ee=b+3+sec-4;int patched=0;
    while(es+4<TS_SZ&&es+4<=ee){
        uint8_t st=pkt[es];
        uint16_t sp=((pkt[es+1]&0x1F)<<8)|pkt[es+2];
        int el=((pkt[es+3]&0xF)<<8)|pkt[es+4];
        if(sp==vid_pid&&st!=0x1B){pkt[es]=0x1B;patched=1;}
        es+=5+el;}
    if(!patched)return 0;
    int coff=b+3+sec-4;if(coff+4>TS_SZ)return 1;
    uint32_t crc=crc32_mpeg(pkt+b,sec-1);
    pkt[coff+0]=(crc>>24)&0xFF;pkt[coff+1]=(crc>>16)&0xFF;
    pkt[coff+2]=(crc>>8)&0xFF; pkt[coff+3]=(crc)&0xFF;
    return 1;}

/* H264 NAL scanner. Scans ALL NALs in the packet payload.
 * Returns IDR(5/20) immediately if found — highest priority.
 * Skips AUD(9) and SEI(6) entirely — they precede every frame
 * and must not mask the real IDR/SPS/PPS that follows them.
 * Returns SPS(7), PPS(8), or other NAL type as fallback.
 * Returns -1 if no H264 NAL found.                           */
static int nal_type(const uint8_t*pkt){
    int off=ts_poff(pkt);if(off<0)return -1;
    const uint8_t*pay=pkt+off;int plen=TS_SZ-off;
    if(ts_pusi(pkt)){
        if(plen<9||pay[0]!=0||pay[1]!=0||pay[2]!=1)return -1;
        int skip=9+pay[8];if(skip>=plen)return -1;
        pay+=skip;plen-=skip;}
    int best=-1;
    for(int i=0;i+3<plen;i++){
        int nt=-1;
        if(pay[i]==0&&pay[i+1]==0&&pay[i+2]==0&&i+4<plen&&pay[i+3]==1)
            {nt=pay[i+4]&0x1F;i+=4;}
        else if(pay[i]==0&&pay[i+1]==0&&pay[i+2]==1&&i+3<plen)
            {nt=pay[i+3]&0x1F;i+=3;}
        if(nt<0)continue;
        if(nt==5||nt==20)return nt;   /* IDR: return immediately */
        if(nt==9||nt==6)continue;     /* AUD/SEI: skip, keep scanning */
        if(nt==7&&best!=5&&best!=20)best=nt;  /* SPS */
        else if(nt==8&&best<0)best=nt;         /* PPS */
        else if(best<0)best=nt;
    }
    return best;}
static int has_idr(const uint8_t*pkt){int n=nal_type(pkt);return n==5||n==20;}

/* Scan a TS packet's payload for the first slice NAL (type 1 or 5) and
 * apply the real Exp-Golomb slice_type check (is_islice) rather than
 * just trusting nal_type()'s raw NAL-type classification. Some real
 * broadcast encoders (confirmed on this exact source family) use
 * ordinary non-IDR NAL units (type 1) for what are functionally
 * keyframes — relying on nal_type()==5 alone means segment cuts almost
 * never fire on such sources, since they essentially never emit a
 * real type-5 IDR. This is the same detection already used in the
 * video_transcode re-encode path (vidxc_push) — applying it here too
 * fixes segment-cut reliability for the simple-copy/audio_transcode-
 * only path on these sources, without needing video_transcode=1.      */
static int has_keyframe(const uint8_t*pkt){
    int off=ts_poff(pkt);if(off<0)return 0;
    const uint8_t*pay=pkt+off;int plen=TS_SZ-off;
    if(ts_pusi(pkt)){
        if(plen<9||pay[0]!=0||pay[1]!=0||pay[2]!=1)return 0;
        int skip=9+pay[8];if(skip>=plen)return 0;
        pay+=skip;plen-=skip;}
    for(int i=0;i+3<plen;i++){
        int nt=-1;int hdr_off=-1;
        if(pay[i]==0&&pay[i+1]==0&&pay[i+2]==0&&i+4<plen&&pay[i+3]==1)
            {nt=pay[i+4]&0x1F;hdr_off=i+4;}
        else if(pay[i]==0&&pay[i+1]==0&&pay[i+2]==1)
            {nt=pay[i+3]&0x1F;hdr_off=i+3;}
        else continue;
        if(nt==1||nt==5){
            int payload_off=hdr_off+1;
            return is_islice(pay+payload_off,plen-payload_off);
        }
    }
    return 0;}

/* Rewrite the first slice NAL (type 1) found in this packet's payload
 * to type 5 (IDR), in place. ONLY called on the exact packet already
 * identified as a cut point via has_keyframe() (i.e. is_islice()
 * already confirmed this slice is genuinely all-intra) — this is a
 * targeted fix for sources that never emit real type-5 IDR NALs (see
 * has_keyframe/is_islice docs for the "functional keyframe" quirk;
 * confirmed via direct inspection of real captured segments: zero NAL
 * type 5 anywhere across 8 consecutive real production segments, only
 * type-1 slices satisfying the Exp-Golomb I-slice check).
 *
 * Rationale: nal_unit_type alone is what tells a decoder "discard all
 * prior reference pictures, this is a clean random-access point" —
 * an Exp-Golomb-confirmed all-intra type-1 slice has the right PICTURE
 * CONTENT for that (no macroblock in it references another frame) but
 * the type-1 tag itself gives the decoder no such instruction, so its
 * reference-picture-buffer bookkeeping (mmco/short-term/long-term ref
 * tracking) carries over from before the segment boundary — exactly
 * what produced the "mmco: unref short failure"/"reference picture
 * missing" errors confirmed on EVERY one of 8 real segments inspected,
 * since any standalone-decoded segment (which is what HLS player-side
 * per-segment decoding effectively is) has no real prior frames for
 * that leftover bookkeeping to refer to. Setting nal_unit_type=5 here
 * is a single bit flip in the NAL header byte (0x01->0x05 in the low
 * 5 bits) and does not touch the slice payload bits at all — the
 * already-intra-coded picture content is unaffected either way.
 *
 * Known residual risk (acknowledged, not fully resolved by this fix):
 * a real decoder seeing nal_unit_type=5 typically also resets its
 * internal frame_num/POC tracking to IDR-relative expectations, while
 * the bitstream's own frame_num field (inside the slice header, not
 * touched by this patch) continues from the prior segment's numbering
 * rather than restarting at the spec-implied baseline for a true IDR.
 * This is a real, unverified-against-a-live-decoder gap — if a
 * regression appears (e.g. a NEW class of artifact right at segment
 * boundaries that wasn't there before), this function is the first
 * thing to disable via idr_rewrite=0.                                 */
static int rewrite_slice_to_idr(uint8_t*pkt){
    int off=ts_poff(pkt);if(off<0)return 0;
    uint8_t*pay=pkt+off;int plen=TS_SZ-off;
    if(ts_pusi(pkt)){
        if(plen<9||pay[0]!=0||pay[1]!=0||pay[2]!=1)return 0;
        int skip=9+pay[8];if(skip>=plen)return 0;
        pay+=skip;plen-=skip;}
    for(int i=0;i+3<plen;i++){
        int nt=-1;int hdr_off=-1;
        if(pay[i]==0&&pay[i+1]==0&&pay[i+2]==0&&i+4<plen&&pay[i+3]==1)
            {nt=pay[i+4]&0x1F;hdr_off=i+4;}
        else if(pay[i]==0&&pay[i+1]==0&&pay[i+2]==1)
            {nt=pay[i+3]&0x1F;hdr_off=i+3;}
        else continue;
        if(nt==1){
            /* Rewrite low 5 bits of the NAL header byte: type 1 -> 5.
             * nal_ref_idc (bits 6-5) and forbidden_zero_bit (bit 7)
             * are left exactly as the source set them.                */
            int ts_byte=(int)(pay-pkt-off)+hdr_off+off;
            if(ts_byte<TS_SZ)
                pkt[ts_byte]=(uint8_t)((pkt[ts_byte]&0xE0)|0x05);
            return 1;
        }
        if(nt==5)return 0; /* already a real IDR, nothing to do */
    }
    return 0;}

/* SPS progressive patch: set frame_mbs_only_flag=1 for LG/Samsung */
static int sps_patch_interlace(uint8_t*pkt){
    int off=ts_poff(pkt);if(off<0)return 0;
    uint8_t*pay=pkt+off;int plen=TS_SZ-off;
    if(ts_pusi(pkt)){
        if(plen<9||pay[0]!=0||pay[1]!=0||pay[2]!=1)return 0;
        int skip=9+pay[8];if(skip>=plen)return 0;
        pay+=skip;plen-=skip;}
    for(int i=0;i+3<plen;i++){
        int n=-1;
        if(pay[i]==0&&pay[i+1]==0&&pay[i+2]==1)n=i+3;
        else if(i+4<plen&&pay[i]==0&&pay[i+1]==0&&pay[i+2]==0&&pay[i+3]==1)n=i+4;
        if(n<0)continue;if(n>=plen)break;
        if((pay[n]&0x1F)!=7){i=n;continue;}
        const uint8_t*sps=pay+n+1;int slen=plen-n-1;
        if(slen<6)break;
        uint8_t profile=sps[0]; int bit=24;
        #define RB(nb)({uint32_t _v=0;for(int _b=0;_b<(nb);_b++){int _y=(bit+_b)/8,_i=7-((bit+_b)%8);if(_y<slen)_v=(_v<<1)|((sps[_y]>>_i)&1);else _v<<=1;}bit+=(nb);_v;})
        #define EG()({int _z=0;while(_z<32){int _y=(bit+_z)/8,_i=7-((bit+_z)%8);if(_y<slen&&!((sps[_y]>>_i)&1))_z++;else break;}bit+=_z+1;uint32_t _v=(1u<<_z)-1;for(int _b=0;_b<_z;_b++){int _y=(bit+_b)/8,_i=7-((bit+_b)%8);if(_y<slen)_v=(_v<<1)|((sps[_y]>>_i)&1);else _v<<=1;}bit+=_z;_v;})
        if(profile==100||profile==110||profile==122||profile==244||
           profile==44||profile==83||profile==86||profile==118||profile==128){
            uint32_t chroma=EG();if(chroma==3)RB(1);
            EG();EG();RB(1);
            if(RB(1)){int nl=(chroma!=3)?8:12;
                for(int sl=0;sl<nl;sl++)
                    if(RB(1)){int sz=(sl<6)?16:64;int last=8,next=8;
                        for(int j=0;j<sz;j++)if(next){int d=(int)EG();next=(last+(d%256)+256)%256;last=next;}}}}
        EG();EG();
        {uint32_t poc=EG();
         if(poc==0){EG();}
         else if(poc==1){RB(1);EG();EG();uint32_t n2=EG();for(uint32_t j=0;j<n2;j++)EG();}}
        EG();RB(1);EG();EG();
        int fb=(bit)/8,fbi=7-(bit%8);
        #undef RB
        #undef EG
        if(fb>=slen)break;
        int ts_byte=(int)(pay-pkt-off)+n+1+fb+off;
        if(ts_byte>=TS_SZ)break;
        if((pkt[ts_byte]>>fbi)&1)break;
        pkt[ts_byte]|=(1<<fbi);
        return 1;}
    return 0;}

static int64_t pes_read_pts(const uint8_t*p){
    return(int64_t)(((uint64_t)(p[0]&0x0E)<<29)|((uint64_t)p[1]<<22)|
                    ((uint64_t)(p[2]&0xFE)<<14)|((uint64_t)p[3]<<7)|
                    ((uint64_t)(p[4]&0xFE)>>1));}
static void pes_write_pts(uint8_t*p,int64_t v,uint8_t m4){
    p[0]=(uint8_t)(m4|((v>>29)&0x0E));p[1]=(uint8_t)((v>>22)&0xFF);
    p[2]=(uint8_t)(0x01|((v>>14)&0xFE));p[3]=(uint8_t)((v>>7)&0xFF);
    p[4]=(uint8_t)(0x01|((v<<1)&0xFE));}
static void pes_patch_pts(uint8_t*pkt,int64_t*base,int*ok){
    if(!ts_pusi(pkt))return;
    int o=ts_poff(pkt);if(o<0||o+8>=TS_SZ)return;
    const uint8_t*pes=pkt+o;
    if(pes[0]!=0||pes[1]!=0||pes[2]!=1)return;
    uint8_t pf=(pes[7]>>6)&3;if(!pf)return;
    int po=o+9;if(po+5>TS_SZ)return;
    int64_t pts=pes_read_pts(pkt+po);
    if(!*ok){*base=pts;*ok=1;}
    int64_t np=pts-*base;if(np<0)np=0;
    pes_write_pts(pkt+po,np,(pf==3)?0x31:0x21);
    if(pf==3){int dp=po+5;if(dp+5<=TS_SZ){
        int64_t dts=pes_read_pts(pkt+dp);
        int64_t nd=dts-*base;if(nd<0)nd=0;
        pes_write_pts(pkt+dp,nd,0x11);}}}

/* ════════════════════════════════════════════════════════════════
   WRITE / SEGMENT / PLAYLIST
   ════════════════════════════════════════════════════════════════ */
static void wb_flush(Ch*c){
    if(!c->wb_n||c->seg_fd<0)return;
    ssize_t r=write(c->seg_fd,c->wb,(size_t)c->wb_n*TS_SZ);(void)r;
    c->wb_n=0;}
static void pkt_write(Ch*c,const uint8_t*p){
    if(c->seg_fd<0)return;
    if(c->aes_on)aes_push(&c->aes,c->seg_fd,p);
    else{memcpy(c->wb+c->wb_n*TS_SZ,p,TS_SZ);if(++c->wb_n>=WRITE_BATCH)wb_flush(c);}}

static void write_m3u8(Ch*c){
    uint64_t complete=c->seg_seq-c->start_seq;
    if(complete<3)return;
    if(!c->target_dur_fixed){c->target_dur=SEGMENT_SECS+2;c->target_dur_fixed=1;}
    uint64_t win=complete>(uint64_t)MAX_SEGMENTS?(uint64_t)MAX_SEGMENTS:complete;
    uint64_t seq0=c->seg_seq-win;
    char tmp[512],fin[512];
    snprintf(fin,512,"%s/%s.m3u8",c->dir,c->name);
    snprintf(tmp,512,"%s/%s.m3u8.tmp",c->dir,c->name);
    FILE*f=fopen(tmp,"w");if(!f)return;
    fprintf(f,"#EXTM3U\n#EXT-X-VERSION:3\n"
              "#EXT-X-TARGETDURATION:%d\n"
              "#EXT-X-MEDIA-SEQUENCE:%llu\n",
              c->target_dur,(unsigned long long)seq0);
    if(c->keyuri[0])
        fprintf(f,"#EXT-X-KEY:METHOD=AES-128,URI=\"%s\",IV=0x%s\n",
                c->keyuri,c->iv_hex);
    for(uint64_t s=seq0;s<c->seg_seq;s++){
        double d=c->dur[s%DUR_SLOTS];
        if(d<=0)d=(double)SEGMENT_SECS;
        if(d>(double)c->target_dur)d=(double)c->target_dur;
        fprintf(f,"#EXTINF:%.3f,\nindex%llu.ts\n",d,(unsigned long long)s);}
    fflush(f);fclose(f);rename(tmp,fin);}

static void seg_close(Ch*c,const struct timespec*now){
    if(c->seg_fd<0)return;
    if(c->aes_on)aes_flush(&c->aes,c->seg_fd,1);else wb_flush(c);
    uint64_t fsize=0;{struct stat st;if(fstat(c->seg_fd,&st)==0)fsize=st.st_size;}
    close(c->seg_fd);c->seg_fd=-1;
    double el=c->seg_timing_ok?wall_el(&c->seg_start,now):0.0;
    uint64_t cl=c->seg_seq-1;
    c->dur[cl%DUR_SLOTS]=el;c->seg_size[cl%DUR_SLOTS]=fsize;c->segs++;
    {uint32_t em=(uint32_t)(el*1000);uint32_t kbps=el>0?(uint32_t)(fsize*8/el/1000):0;
     Evt e={EVT_CLOSE,time(NULL),cl,em,kbps,0,0,0};evpush(&c->evring,&e);}
    write_m3u8(c);}

/* Abort a segment damaged by a CC drop without publishing it.
 *
 * This is the simple-copy-mode alternative to seg_close() on CC error.
 * The difference: seg_close() calls write_m3u8(), which immediately
 * publishes the segment to the player's playlist — so a short, damaged
 * segment (e.g. 0.059s, confirmed from real logs) becomes visible to
 * ExoPlayer, which then tries to decode a truncated or mid-GOP
 * bitstream and freezes or shows macroblocks until the next clean IDR
 * arrives. seg_abort() instead:
 *
 *   1. Flushes and closes the file (so OS resources are released cleanly)
 *   2. Deletes the file from disk (no partial segment left behind)
 *   3. Rolls back seg_seq by 1 (so the next seg_open() reuses this slot)
 *   4. Does NOT call write_m3u8() (player never sees this segment)
 *   5. Does NOT increment c->segs (doesn't count as a produced segment)
 *
 * The net effect from the player's perspective: the previous good
 * segment is the last thing it sees, followed eventually by a new
 * clean segment starting on an IDR. No broken intermediate segment
 * appears in the playlist at all. The player may briefly stall waiting
 * for the next segment (the recovery time until an IDR arrives — at
 * 5.5Mbps with a 2s GOP, typically 0-2s), but a brief stall is
 * dramatically better than a freeze-on-broken-segment or macroblock
 * burst that ExoPlayer can take several seconds to recover from.
 *
 * Only used for simple-copy mode (video_transcode=0). The transcode
 * path already has its own CC recovery via the encoder's own IDR
 * re-emission cycle, which handles this differently.                  */
static void seg_abort(Ch*c){
    if(c->seg_fd<0)return;
    /* Flush+close without calling write_m3u8 */
    if(c->aes_on)aes_flush(&c->aes,c->seg_fd,1);else wb_flush(c);
    close(c->seg_fd);c->seg_fd=-1;
    /* Delete the partial file — don't leave a broken segment on disk */
    char path[512];
    snprintf(path,512,"%s/index%llu.ts",c->dir,
             (unsigned long long)(c->seg_seq-1));
    unlink(path);
    /* Roll back seg_seq so the next seg_open() reuses this slot number.
     * This keeps the playlist sequence gap-free: the aborted segment
     * slot is simply never published, and the next clean segment takes
     * its place as if the abort never happened.                        */
    c->seg_seq--;
    c->seg_timing_ok=0;
    /* Log the abort so it's visible in the event ring / stats output,
     * distinct from a normal seg_close. Uses EVT_CLOSE with duration=0
     * as a proxy since there's no dedicated EVT_ABORT type — the 0ms
     * duration distinguishes it from a real close in post-analysis.   */
    {Evt e={EVT_CLOSE,time(NULL),c->seg_seq,0,0,0,0,0};
     evpush(&c->evring,&e);}}

static void seg_open_ex(Ch*c,const struct timespec*now,int suppress_replay){
    int keep=MAX_SEGMENTS+g_del;
    if(c->seg_seq>=(uint64_t)keep+c->start_seq){
        char dp[512];
        snprintf(dp,512,"%s/index%llu.ts",c->dir,
                 (unsigned long long)(c->seg_seq-(uint64_t)keep));
        unlink(dp);}
    char path[512];
    snprintf(path,512,"%s/index%llu.ts",c->dir,(unsigned long long)c->seg_seq);
    c->seg_fd=open(path,O_WRONLY|O_CREAT|O_TRUNC,0644);
    if(c->seg_fd<0)return;
    if(c->aes_on)aes_reset(&c->aes);
    c->wb_n=0;c->seg_seq++;c->seg_start=*now;c->seg_timing_ok=1;
    /* Write patched PAT+PMT then cached SPS+PPS — unless the caller
     * already knows the very next packet it writes supplies fresh
     * SPS+PPS itself (suppress_replay), in which case replaying the
     * cache here would produce a duplicate parameter-set sequence at
     * the segment boundary (see call site in the segment-cut block
     * for the full rationale and the real corruption this caused).   */
    if(c->pat_ok)pkt_write(c,c->pat);
    if(c->pmt_ok)pkt_write(c,c->pmt);
    if(!suppress_replay)
        for(int i=0;i<c->spspps_n;i++)pkt_write(c,c->spspps[i]);
    {Evt e={EVT_OPEN,time(NULL),c->seg_seq-1,0,0,0,0,0};evpush(&c->evring,&e);}}
static void seg_open(Ch*c,const struct timespec*now){
    seg_open_ex(c,now,0);}

/* ════════════════════════════════════════════════════════════════
   PACKET PROCESSOR
   ════════════════════════════════════════════════════════════════ */
static int i64cmp(const void*a,const void*b){
    int64_t x=*(const int64_t*)a,y=*(const int64_t*)b;
    return (x>y)-(x<y);}
static int64_t igcd(int64_t a,int64_t b){while(b){int64_t t=b;b=a%b;a=t;}return a?a:1;}

/* fps auto-detection: collects raw video-PID pts deltas and, once 16
 * samples are in, derives fpn/fpd from their median (see the Ch struct
 * field doc above for the full rationale). 16 samples is enough margin
 * to get a stable estimate within well under a second of real time for
 * any framerate likely to show up here (worst case ~25fps -> ~17
 * access units -> well under 700ms), so this doesn't meaningfully delay
 * video_transcode startup. Falls back to the old default (25/1) if
 * deltas never settle to a sane positive value (e.g. a channel with a
 * genuinely broken/absent pts on every packet) -- never blocks init
 * forever.                                                            */
/* Read-only twin of sps_patch_interlace() -- identical SPS bit parsing,
 * but never writes to the packet. Returns 1 if frame_mbs_only_flag==1
 * (progressive), 0 if ==0 (genuinely interlaced), -1 if no SPS found
 * in this packet or parsing ran out of bits before reaching the flag.
 * Needed because raw PES-to-PES pts timing alone can't distinguish
 * "genuinely 50fps progressive" from "genuinely 25fps interlaced, but
 * the source PES-packages each field separately" -- both look like a
 * ~1800-tick access-unit delivery rate on the wire. frame_mbs_only_flag
 * is the actual unambiguous bitstream signal for which one it really
 * is, straight from the SPS itself, not a timing heuristic.            */
static int sps_read_frame_mbs_only(const uint8_t*pkt){
    int off=ts_poff(pkt);if(off<0)return -1;
    const uint8_t*pay=pkt+off;int plen=TS_SZ-off;
    if(!ts_pusi(pkt))return -1;
    if(plen<9||pay[0]!=0||pay[1]!=0||pay[2]!=1)return -1;
    int skip=9+pay[8];if(skip>=plen)return -1;
    pay+=skip;plen-=skip;
    for(int i=0;i+3<plen;i++){
        int n=-1;
        if(pay[i]==0&&pay[i+1]==0&&pay[i+2]==1)n=i+3;
        else if(i+4<plen&&pay[i]==0&&pay[i+1]==0&&pay[i+2]==0&&pay[i+3]==1)n=i+4;
        if(n<0)continue;if(n>=plen)break;
        if((pay[n]&0x1F)!=7){i=n;continue;}
        const uint8_t*sps=pay+n+1;int slen=plen-n-1;
        if(slen<6)break;
        uint8_t profile=sps[0]; int bit=24;
        #define RB(nb)({uint32_t _v=0;for(int _b=0;_b<(nb);_b++){int _y=(bit+_b)/8,_i=7-((bit+_b)%8);if(_y<slen)_v=(_v<<1)|((sps[_y]>>_i)&1);else _v<<=1;}bit+=(nb);_v;})
        #define EG()({int _z=0;while(_z<32){int _y=(bit+_z)/8,_i=7-((bit+_z)%8);if(_y<slen&&!((sps[_y]>>_i)&1))_z++;else break;}bit+=_z+1;uint32_t _v=(1u<<_z)-1;for(int _b=0;_b<_z;_b++){int _y=(bit+_b)/8,_i=7-((bit+_b)%8);if(_y<slen)_v=(_v<<1)|((sps[_y]>>_i)&1);else _v<<=1;}bit+=_z;_v;})
        if(profile==100||profile==110||profile==122||profile==244||
           profile==44||profile==83||profile==86||profile==118||profile==128){
            uint32_t chroma=EG();if(chroma==3)RB(1);
            EG();EG();RB(1);
            if(RB(1)){int nl=(chroma!=3)?8:12;
                for(int sl=0;sl<nl;sl++)
                    if(RB(1)){int sz=(sl<6)?16:64;int last=8,next=8;
                        for(int j=0;j<sz;j++)if(next){int d=(int)EG();next=(last+(d%256)+256)%256;last=next;}}}}
        EG();EG();
        {uint32_t poc=EG();
         if(poc==0){EG();}
         else if(poc==1){RB(1);EG();EG();uint32_t n2=EG();for(uint32_t j=0;j<n2;j++)EG();}}
        EG();RB(1);EG();EG();
        int fb=(bit)/8;
        #undef RB
        #undef EG
        if(fb>=slen)return -1;
        int ts_byte=(int)(pay-pkt-off)+n+1+fb+off;
        if(ts_byte>=TS_SZ)return -1;
        int fbi=7-(bit%8);
        return (pkt[ts_byte]>>fbi)&1;
    }
    return -1;
}

static void fps_probe_finalize_apply(Ch*c,int fpn,int fpd,int64_t rep,
                                      const int64_t*tmp);

static void fps_probe_sample(Ch*c,int64_t pts){
    if(!c->fps_probe_have_last){
        c->fps_probe_last_pts=pts;c->fps_probe_have_last=1;return;}
    int64_t d=pts-c->fps_probe_last_pts;
    c->fps_probe_last_pts=pts;
    if(d<450||d>90000*2)return; /* skip implausibly short (<450 ticks,
        i.e. faster than 200fps -- no real broadcast source goes there,
        so this can only be a glitched/duplicate-packet artifact) or
        implausibly long (>2s, e.g. right after a STALE recovery) gaps.
        The lower bound matters more now than it used to: since the
        raw minimum is used directly with no population/clustering
        requirement diluting a single bad sample, a stray near-zero
        glitch could otherwise become the answer outright.             */
    if(c->fps_probe_n<16)c->fps_probe_deltas[c->fps_probe_n++]=d;
    if(c->fps_probe_n>=16){
        int64_t tmp[16];memcpy(tmp,c->fps_probe_deltas,sizeof tmp);
        qsort(tmp,16,sizeof(int64_t),i64cmp);
        /* Use the GCD of all 16 raw deltas, not just the minimum, and
         * not clustering, and not a plain median. History of what
         * didn't work, for anyone revisiting this:
         *
         * Plain median: lands in the GAP between two genuinely-
         * occurring delta clusters (confirmed real case: a 50fps
         * stream where ~half the measured deltas come out doubled) --
         * produced fps=100/3 (33.333), not a real broadcast rate.
         *
         * Largest cluster: also wrong on a different real channel --
         * an 8/16 vs 8/16 split where the larger-delta cluster (pairs
         * of missed access units) won, producing fps=25/2 (12.5).
         *
         * Smallest cluster requiring >=3 members: STILL wrong on a
         * third real channel -- deltas were {3600,3600,7200x8,14400x2,
         * 18000x3,21600}, every value an exact multiple of 3600, with
         * the true period appearing only twice. The >=3 threshold
         * discarded the correct answer for being rare in that window.
         *
         * Raw minimum (no threshold at all): STILL wrong on a FOURTH
         * real channel -- deltas were {7200x8,10800x4,32400x4}. The
         * true 1x period (3600) never appeared ANYWHERE in this
         * window -- every single access unit in this 16-sample capture
         * skipped at least one PES header, so "take the smallest
         * observed value" could only ever report 7200 (2x), since
         * that's genuinely the smallest thing that showed up.
         *
         * What actually recovers the true period even when it was
         * never directly sampled: the GCD of all 16 deltas. Every
         * value in every case above is an exact integer multiple of
         * the true period (never a fraction of it -- a missed PES
         * header can only skip whole access units), so GCD reconstructs
         * it correctly regardless of which particular multiples
         * happened to show up in any given 16-sample window. Real
         * broadcast video pts is exact-integer-tick precise (not
         * subject to the kind of fractional jitter that would make
         * GCD unstable the way it can be for e.g. resampled audio), so
         * plain integer GCD is safe here. Re-verified against all four
         * real bugs above plus every clean case (25/50/29.97fps, and a
         * mostly-50-with-outliers case) before shipping this -- all
         * correct, including the case that just broke raw-minimum.     */
        int64_t rep=tmp[0];
        for(int i=1;i<16;i++)rep=igcd(rep,tmp[i]);
        int chosen_start=0,chosen_len=1;
        double measured_fps=90000.0/(double)rep;
        /* Snap to the nearest standard broadcast rate when close enough
         * -- a second, independent guard against the clustering result
         * still landing slightly off a real rate (jitter, rounding).
         * Falls back to the raw gcd-reduced measurement, not a
         * hardcoded default, if nothing standard is close -- so a
         * genuinely unusual-but-real rate still gets honored rather
         * than forced onto the nearest standard one.                  */
        static const struct{int fpn,fpd;double fps;}STD[]={
            {24000,1001,23.976},{24,1,24.0},{25,1,25.0},
            {30000,1001,29.97},{30,1,30.0},{50,1,50.0},
            {60000,1001,59.94},{60,1,60.0}};
        int fpn=0,fpd=0;double bestdiff=1e9;
        for(int i=0;i<8;i++){
            double diff=measured_fps-STD[i].fps;if(diff<0)diff=-diff;
            if(diff<bestdiff){bestdiff=diff;fpn=STD[i].fpn;fpd=STD[i].fpd;}
        }
        if(bestdiff>0.75){
            fpn=90000;fpd=(int)rep;
            int64_t g=igcd(fpn,fpd);fpn/=g;fpd/=g;
        }
        /* Interlace correction: raw PES-to-PES timing can't tell "50fps
         * progressive" apart from "25fps interlaced, PES-packaged per
         * field" -- both put a genuine ~1800-tick access-unit on the
         * wire. Confirmed via real production log: a 720x576 (PAL SD,
         * commonly interlaced) source measured as 50/1 by timing alone,
         * but is actually 25fps. frame_mbs_only_flag straight from the
         * SPS is the actual unambiguous signal for which one it is.
         * SPS only arrives once per GOP/IDR though, not every packet,
         * so it may well not have shown up inside just the 16-sample
         * timing window yet -- when the timing result lands on one of
         * the field/frame-ambiguous rates and mbs_only is still
         * unknown, defer finalizing (don't set fps_probe_done) so the
         * caller keeps scanning packets specifically for an SPS beyond
         * this point, instead of locking in a possibly-wrong guess.    */
        int ambiguous=(fpn==50&&fpd==1)||(fpn==60&&fpd==1)||
                       (fpn==60000&&fpd==1001);
        if(c->fps_probe_mbs_only==0){
            if(fpn==50&&fpd==1){fpn=25;fpd=1;}
            else if(fpn==60&&fpd==1){fpn=30;fpd=1;}
            else if(fpn==60000&&fpd==1001){fpn=30000;fpd=1001;}
        } else if(ambiguous&&c->fps_probe_mbs_only<0&&
                  c->fps_probe_extra_pkts<200){
            if(!c->fps_probe_ambiguous){
                c->fps_probe_pending_fpn=fpn;c->fps_probe_pending_fpd=fpd;
                c->fps_probe_pending_rep=rep;
                printf("[%s] video_transcode: fps probe landed on an "
                       "ambiguous rate (%d/%d) with no SPS read yet -- "
                       "extending search for frame_mbs_only_flag before "
                       "finalizing\n",c->name,fpn,fpd);
            }
            c->fps_probe_ambiguous=1;
            (void)chosen_start;(void)chosen_len;
            return; /* deferred -- see fps_probe_try_finalize_deferred */
        }
        fps_probe_finalize_apply(c,fpn,fpd,rep,tmp);
        (void)chosen_start;(void)chosen_len;
    }
}

/* Applies the safety-net halving (if still needed), the sanity backstop,
 * and marks the probe done -- shared by the immediate-finalize path
 * above and the deferred/extended-search path below, so both produce
 * identically-corrected results.                                       */
static void fps_probe_finalize_apply(Ch*c,int fpn,int fpd,int64_t rep,
                                      const int64_t*tmp){
    /* Safety net: this deployment is exclusively PAL-region -- every
     * legitimate source here is 25 or 50fps, never a genuine 50/60fps
     * progressive feed. If the extended SPS search above still
     * couldn't get a definitive frame_mbs_only_flag (source never
     * repeated an SPS within the 200-packet extension window) and the
     * timing result is one of the field/frame-ambiguous rates, treat
     * it as interlaced and halve anyway rather than trusting an
     * unconfirmed guess -- a wrongly-halved genuine progressive 50fps
     * source (which doesn't exist in this deployment) would just look
     * subtly soft; a wrongly-NOT-halved interlaced source (the real,
     * observed failure) desyncs audio/video permanently for the whole
     * channel run, so this is the safer side to default to.            */
    if(c->fps_probe_mbs_only<0){
        if(fpn==50&&fpd==1){
            fpn=25;fpd=1;
            printf("[%s] video_transcode: SPS never confirmed "
                   "interlace/progressive -- applying safety-net "
                   "halving 50/1 -> 25/1 (PAL-only deployment)\n",c->name);
        } else if(fpn==60&&fpd==1){
            fpn=30;fpd=1;
            printf("[%s] video_transcode: SPS never confirmed "
                   "interlace/progressive -- applying safety-net "
                   "halving 60/1 -> 30/1 (PAL-only deployment)\n",c->name);
        } else if(fpn==60000&&fpd==1001){
            fpn=30000;fpd=1001;
            printf("[%s] video_transcode: SPS never confirmed "
                   "interlace/progressive -- applying safety-net "
                   "halving 60000/1001 -> 30000/1001 (PAL-only "
                   "deployment)\n",c->name);
        }
    }
    /* Final sanity backstop, floor raised to 25fps specifically for
     * this deployment: every channel here is PAL-region (25/50fps
     * sources) -- there is no legitimate content expected below
     * 25fps at all, unlike a mixed NTSC/film deployment where a
     * genuine 23.976fps detection would need to survive this check.
     * So rather than only catching wildly-implausible values (the
     * original 10-120fps range), clamp ANY result under 25fps
     * straight to 25/1 -- this closes off every variant of the
     * "detector silently lands on a too-low rate" bug class at
     * once, not just the specific raw-delta patterns already found
     * and fixed above, in case some future pattern this deployment
     * hasn't hit yet would otherwise slip through GCD too.
     * Upper bound stays generous (120fps) since over-detecting
     * isn't the failure mode that's ever actually been observed.   */
    double final_fps=(double)fpn/fpd;
    if(final_fps<25.0||final_fps>120.0){
        fprintf(stderr,"[%s] video_transcode: fps probe produced an "
                "implausible result (%.3f fps) -- falling back to "
                "25/1\n",c->name,final_fps);
        fpn=25;fpd=1;
    }
    c->fps_detected_fpn=fpn;c->fps_detected_fpd=fpd;
    c->fps_probe_done=1;
    /* The decoder (VidXc) may already exist by now -- it's no longer
     * gated on this probe finishing (see the fix at the vidxc creation
     * call site). Its encoder, however, IS gated on fps_confirmed (see
     * vidxc_push), so it cannot have opened yet with a wrong/default
     * fps -- there's nothing to tear down or reopen here, just confirm
     * the real value so the encoder opens correctly whenever the next
     * decoded frame does trigger vidxc_open_enc(). This replaces an
     * earlier version of this fix that let the encoder open eagerly on
     * whatever fps was available and detected+repaired a mismatch
     * after the fact (confirmed necessary once: a real 1920x1080 50fps
     * source decoded its first frame fast enough to open the encoder
     * at the 25/1 default before this probe finished) -- gating the
     * encoder open itself is strictly better: no wasted encoder
     * open/close cycle, no dropped GOP restart, same small bounded
     * wait either way.                                                 */
    if(c->vidxc&&c->vidxc->active){
        c->vidxc->fpn=fpn;c->vidxc->fpd=fpd;
        c->vidxc->fps_confirmed=1;
    }
    /* Dump every raw sample, sorted -- so a wrong result can be
     * diagnosed directly from the log instead of guessing whether
     * it's a genuinely different sample distribution or a stale
     * binary (this is what actually caught every clustering bug
     * above -- keep it).                                           */
    printf("[%s] video_transcode: fps probe raw deltas (ticks):",c->name);
    for(int i=0;i<16;i++)printf(" %lld",(long long)tmp[i]);
    printf("\n");
    printf("[%s] video_transcode: auto-detected fps=%d/%d "
           "(%.3f fps, gcd=%lld ticks of 16 samples, "
           "mbs_only=%d, extra_pkts=%d)\n",
           c->name,fpn,fpd,(double)fpn/fpd,(long long)rep,
           c->fps_probe_mbs_only,c->fps_probe_extra_pkts);
}

/* Called once per qualifying video-PID PUSI packet, only while a probe
 * is in the deferred/ambiguous state (fps_probe_ambiguous set, done not
 * yet set) -- keeps counting extra packets and re-checks whether
 * mbs_only has resolved yet or the extension cap has been hit, and
 * finalizes as soon as either happens.                                 */
static void fps_probe_try_finalize_deferred(Ch*c){
    if(!c->fps_probe_ambiguous||c->fps_probe_done)return;
    c->fps_probe_extra_pkts++;
    if(c->fps_probe_mbs_only>=0||c->fps_probe_extra_pkts>=200){
        int64_t tmp[16];memcpy(tmp,c->fps_probe_deltas,sizeof tmp);
        qsort(tmp,16,sizeof(int64_t),i64cmp);
        int fpn=c->fps_probe_pending_fpn,fpd=c->fps_probe_pending_fpd;
        int64_t rep=c->fps_probe_pending_rep;
        if(c->fps_probe_mbs_only==0){
            if(fpn==50&&fpd==1){fpn=25;fpd=1;}
            else if(fpn==60&&fpd==1){fpn=30;fpd=1;}
            else if(fpn==60000&&fpd==1001){fpn=30000;fpd=1001;}
        }
        fps_probe_finalize_apply(c,fpn,fpd,rep,tmp);
    }
}

static void pkt_process(Ch*c,const uint8_t*p,const struct timespec*now){
    if(p[0]!=0x47)return;
    uint16_t pid=ts_pid(p);
    if(pid==0x1FFF)return;
    if(c->n_aud_dropped&&aud_is_dropped(c,pid))return; /* audio_map: this
        track was explicitly excluded — drop its packets entirely, not
        just its PMT entry (see aud_is_dropped doc for full rationale) */
    /* fps auto-detect sampling -- cheap no-op once fps_probe_done is
     * set, and only runs at all while video_transcode is configured.
     * Reads pts directly off the raw incoming packet; doesn't need or
     * touch the PES bytes otherwise, so this can't affect anything
     * downstream regardless of what dispatch branch ends up handling
     * this packet.                                                    */
    if(c->opts.video_transcode&&c->vid_pid&&pid==c->vid_pid&&
       !c->fps_probe_done&&ts_pusi(p)){
        int o=ts_poff(p);
        if(o>=0&&o+8<TS_SZ){
            const uint8_t*pes=p+o;
            if(pes[0]==0&&pes[1]==0&&pes[2]==1){
                uint8_t pf=(pes[7]>>6)&3;
                int po=o+9;
                if(pf&&po+5<=TS_SZ){
                    int64_t pts=pes_read_pts(p+po);
                    fps_probe_sample(c,pts);
                }
            }
        }
        /* Try to read frame_mbs_only_flag from this same packet's SPS
         * (if it carries one) until we get a definitive answer -- SPS
         * doesn't appear on every packet, only when actually present,
         * so this may take a few tries across the probe window.        */
        if(c->fps_probe_mbs_only<0){
            int r=sps_read_frame_mbs_only(p);
            if(r>=0)c->fps_probe_mbs_only=r;
        }
        /* If the 16-sample timing probe already landed on an ambiguous
         * rate and is waiting on this extended search (fps_probe_
         * ambiguous set, fps_probe_done still 0), keep counting packets
         * and re-check whether we can finalize now -- either because
         * mbs_only just resolved above, or because the extension cap
         * has been reached and the safety-net halving should apply.    */
        if(c->fps_probe_ambiguous&&!c->fps_probe_done)
            fps_probe_try_finalize_deferred(c);
    }
    c->pkts++;c->snap_bytes+=TS_SZ;
    /* Track last PCR for gap-stuffing interpolation. Updated here,
     * before CC-drop handling, so we always have the most recent
     * 27MHz PCR available when a gap is detected just below.        */
    if(c->pcr_pid&&pid==c->pcr_pid){
        int64_t pcr=ts_read_pcr(p);
        if(pcr>=0)c->last_pcr27=pcr;}

    /* ── PAT ────────────────────────────────────────────────── */
    if(pid==0x0000){
        if(!c->pat_ok){
            uint16_t pp=pat_parse(p);
            if(pp){c->pmt_pid=pp;c->pat_ok=1;
                Evt e={EVT_PAT,time(NULL),0,0,0,pp,0,0};evpush(&c->evring,&e);}}
        if(c->pat_ok)memcpy(c->pat,p,TS_SZ);
        /* Write patched PAT (not raw source) */
        if(c->pat_ok&&c->started)pkt_write(c,c->pat);
        return;}

    /* ── PMT ────────────────────────────────────────────────── */
    if(c->pat_ok&&pid==c->pmt_pid){
        uint16_t vp=0,ap_unused=0,cp=0;uint8_t vst=0,aud_st_unused=0;
        AudTrack disc[MAX_AUD_TRACKS];
        int n_disc=pmt_parse_all_audio(p,disc);
        /* pmt_parse still used for video PID + PCR PID — those remain
         * single-value by nature (a program has one PCR PID and, for
         * our purposes, one video PID), so no need to touch that path. */
        if(pmt_parse(p,&cp,&vp,&ap_unused,&vst,&aud_st_unused)&&cp){
            if(!c->pmt_ok||cp!=c->pcr_pid||vp!=c->vid_pid){
                c->pcr_pid=cp;c->vid_pid=vp;c->vid_stream_type=vst;c->pmt_ok=1;
                /* Determine which discovered audio track(s) to select. */
                int sel_lo=0,sel_hi=n_disc; /* default: select ALL tracks */
                if(c->opts.audio_map>0){
                    if(c->opts.audio_map<=n_disc){sel_lo=c->opts.audio_map-1;sel_hi=c->opts.audio_map;}
                    else{sel_lo=sel_hi=0; /* requested index doesn't exist */
                        fprintf(stderr,"[%s] WARNING: audio_map=%d requested but "
                                "only %d audio track(s) found in PMT — no audio "
                                "track selected.\n",c->name,c->opts.audio_map,n_disc);}
                }
                c->n_aud_active=0;c->n_aud_dropped=0;
                for(int i=0;i<n_disc;i++){
                    int selected=(i>=sel_lo&&i<sel_hi);
                    if(!selected){
                        if(c->opts.audio_map>0&&c->n_aud_dropped<MAX_AUD_TRACKS)
                            c->aud_dropped[c->n_aud_dropped++]=disc[i].pid;
                        continue;
                    }
                    if(c->n_aud_active>=MAX_AUD_TRACKS)continue;
                    AudSlot*s=&c->aud[c->n_aud_active];
                    s->src_pid=s->out_pid=disc[i].pid;
                    s->stream_type=disc[i].stream_type;
                    c->n_aud_active++;
                    if(c->opts.audio_transcode){
                        if(!is_decodable_audio(disc[i].stream_type))
                            fprintf(stderr,"[%s] WARNING: audio_transcode requested but "
                                    "audio track %d (PID 0x%04X) has stream_type=0x%02X "
                                    "(not MP2/MP3/AAC) — our decoder cannot handle this; "
                                    "this track will be passed through untouched.\n",
                                    c->name,i+1,disc[i].pid,disc[i].stream_type);
                        else if(disc[i].stream_type==0x0F)
                            printf("[%s] audio track %d (PID 0x%04X) is already AAC — "
                                   "audio_transcode skipped for it, passing through as-is\n",
                                   c->name,i+1,disc[i].pid);
                    }
                }
                Evt e={EVT_PMT,time(NULL),0,(uint32_t)(c->n_aud_active?c->aud[0].src_pid:0),
                       (uint32_t)cp,vp,0,0};
                evpush(&c->evring,&e);}}
        /* Cache and patch PMT — NEVER write raw source PMT to segments.
         * Patch stream_type for every track we'll transcode to AAC, and
         * (for audio_map mode) strip any audio track NOT selected, per
         * the design: unselected tracks are dropped entirely from the
         * output, not merely left untranscoded.                        */
        memcpy(c->pmt,p,TS_SZ);
        if(c->opts.audio_map>0)
            pmt_strip_unselected_audio(c->pmt,c->n_aud_active?c->aud[0].src_pid:0);
        if(c->opts.fix_mp2||c->opts.audio_transcode)
            for(int i=0;i<c->n_aud_active;i++)
                pmt_patch_audio(c->pmt,c->aud[i].src_pid);
        if(c->opts.video_transcode&&c->vid_pid)
            pmt_patch_video(c->pmt,c->vid_pid);
        /* Init audio transcoder for each selected track that needs one
         * (genuinely decodable source — MP2/MP3 via mpg123, or AC3/
         * E-AC3 via libavcodec; already-AAC tracks are passed through
         * with no transcoder, per the audxc.active guard in the
         * per-packet dispatch below).                                  */
        if(c->opts.audio_transcode)
            for(int i=0;i<c->n_aud_active;i++){
                AudSlot*s=&c->aud[i];
                if(s->stream_type==0x0F||!is_decodable_audio(s->stream_type))continue;
                if(s->audxc.active)continue;
                AudXcSrc src_codec=(s->stream_type==0x06||s->stream_type==0x87||
                                     s->stream_type==0x81)?
                                    AUDXC_SRC_AC3:AUDXC_SRC_MP2;
                printf("[%s] audio_transcode: track%d src_pid=0x%04X codec=%s "
                       "bitrate=%d coder=%s\n",
                       c->name,i+1,s->src_pid,
                       src_codec==AUDXC_SRC_AC3?"AC3->AAC":"MP2->AAC",
                       c->opts.audio_bitrate>0?c->opts.audio_bitrate:128000,
                       c->opts.audio_coder);
                if(!audxc_init(&s->audxc,c->opts.audio_bitrate,c->opts.audio_coder,src_codec))
                    fprintf(stderr,"[%s] audxc_init FAILED for track%d\n",c->name,i+1);
                s->cc=0;s->audxc.pusi_seen=0;}
        /* Init video transcoder once -- VidXc is heap-allocated here,
         * lazily, only for channels that actually set video_transcode=1
         * (see the VidXc* field doc in Ch for why).
         * FIX: this used to also gate on c->fps_probe_done, deferring
         * creation until the fps auto-detect probe finished (up to 16
         * samples, or up to ~200 with the ambiguous-rate extension).
         * That's wrong: vidxc_init() only opens the DECODER, which
         * doesn't need fps at all (it defaults v->fpn/fpd to 25/1 if
         * not yet known) -- the ENCODER, which genuinely does need
         * fps for time_base/gop_size, is already opened separately and
         * lazily inside vidxc_push() on the first successfully decoded
         * frame, independent of when the decoder was created. Gating
         * decoder creation on the fps probe meant any IDR arriving
         * during that probe window was silently dropped (no VidXc
         * existed yet to receive it), forcing a wait for the NEXT IDR
         * before the channel could ever open its first segment --
         * confirmed as the cause of freeze/frame-stop reports on
         * channels whose first IDR happened to land inside the probe
         * window. Now the decoder opens immediately once vid_pid is
         * known (same timing as audio init), using whatever fps is
         * available at that instant (0 -> defaults to 25/1 inside
         * vidxc_init, harmless since it's not used until the encoder
         * opens); fps_probe_finalize_apply refreshes v->fpn/fpd with
         * the real detected value before the encoder actually opens,
         * if the probe finishes after decoder creation but before the
         * first frame is decoded (see the sync there).                 */
        if(c->opts.video_transcode&&c->vid_pid&&!c->vidxc){
            c->vidxc=calloc(1,sizeof(VidXc));
            if(!c->vidxc){
                fprintf(stderr,"[%s] video_transcode: out of memory\n",c->name);
            } else {
                uint8_t st=c->vid_stream_type?c->vid_stream_type:0x1B;
                float crf=c->opts.video_crf>0?c->opts.video_crf:23.0f;
                int fpn=c->fps_detected_fpn,fpd=c->fps_detected_fpd;
                if(vidxc_init(c->vidxc,st,fpn,fpd,crf))
                    printf("[%s] video_transcode: 0x%02X crf=%.0f fps=%d/%d\n",
                           c->name,st,crf,fpn,fpd);
                else {
                    /* init failed -- free immediately so we retry
                     * cleanly next time this block is reached rather
                     * than leaking a half-initialized struct forever. */
                    free(c->vidxc); c->vidxc=NULL;
                }
            }
        }
        /* Write patched PMT to segment */
        if(c->started)pkt_write(c,c->pmt);
        return;}

    /* ── CC drop detection ──────────────────────────────────── */
    /* Only video/audio PIDs trigger seg_close+recovery — a CC blip on
     * some other PID (SDT, EIT, or any other metadata/PSI PID we don't
     * even forward into the output) is completely harmless to playback
     * since we never write those packets to a segment anyway. Treating
     * every PID's CC discontinuity as disruptive caused frequent,
     * spurious segment truncation/resync on real broadcast sources —
     * confirmed via a 70s continuous test: a periodic CC blip on PID
     * 0x0011 (SDT) was forcing seg_close()+recovering=1 roughly every
     * ~10s, producing truncated segments (as short as 0.348s) and a
     * brief resync gap each time — closely matching reports of the
     * stream "stopping every 10 seconds" in VLC and on an LG TV. We
     * still count/track CC errors on every PID for stats visibility,
     * but only act on them for PIDs that actually end up in the output.*/
    {int afc=ts_afc(p);
     if((afc==1||afc==3)&&pid!=0x1FFF&&pid!=0x0000&&pid!=c->pmt_pid){
         uint8_t cc=(uint8_t)(p[3]&0x0F);
         uint8_t prev=c->cc_last[pid&0x1FFF];
         if(prev!=0xFF){
             uint8_t exp=(uint8_t)((prev+1)&0x0F);
             if(cc!=exp&&cc!=prev){
                 c->cc_errors++;
                 int relevant=(pid==c->vid_pid)||(aud_slot_for_pid(c,pid)>=0);
                 if(relevant&&c->started){
                     Evt e={EVT_CC,time(NULL),0,0,0,(uint16_t)(pid&0x1FFF),exp,cc};
                     evpush(&c->evring,&e);
                     /* For a VIDEO-PID drop under video_transcode, decide
                      * the cooldown ONCE and gate seg_close together with
                      * the decoder-reset actions below — otherwise the
                      * segment still gets torn down on every drop in a
                      * burst even though the decoder reset is correctly
                      * throttled, and reopening is gated on the (now
                      * recovering) decoder producing a keyframe, so the
                      * gap persists for the same reason just one step
                      * removed. Confirmed via direct re-analysis of real
                      * production logs taken AFTER the first cooldown
                      * attempt: the same multi-second gaps continued,
                      * because this exact call was still unconditional.
                      * Audio-only drops (or video drops when
                      * video_transcode isn't active) keep the original
                      * always-close behavior — the cooldown is
                      * specifically about not interrupting an in-progress
                      * VIDEO recovery, since video decode/keyframe
                      * detection is the real bottleneck, not segment
                      * file I/O.                                        */
                     int video_recovery_throttled=0;
                     if(c->opts.video_transcode&&pid==c->vid_pid){
                         double since_last=c->last_recovery_trigger_valid?
                             wall_el(&c->last_recovery_trigger,now):1e9;
                         video_recovery_throttled=(since_last<0.7);
                     }
                     if(c->seg_fd>=0&&!video_recovery_throttled){
                         /* Simple-copy mode: abort the damaged segment
                          * rather than closing it normally. seg_abort()
                          * deletes the partial file and rolls back
                          * seg_seq without calling write_m3u8() — so
                          * ExoPlayer never sees this broken segment in
                          * the playlist at all. The player gets a brief
                          * stall (time until the next clean IDR) instead
                          * of the current behaviour (a very short,
                          * mid-GOP segment → freeze/macroblock that can
                          * persist for several seconds). Confirmed root
                          * cause of the drops: disk I/O stalls on /dev/
                          * sda at 98.4% utilisation causing kernel UDP
                          * receive-buffer overflows (158M errors seen in
                          * netstat -su), dropping entire 1316-byte UDP
                          * datagrams (7 TS packets at once — exactly
                          * matching the observed CC gap sizes of 6-7).
                          * For video_transcode mode, the original
                          * seg_close() is kept since that path manages
                          * its own recovery via the encoder's own IDR
                          * re-emission cycle rather than this state
                          * machine.                                     */
                         /* Simple-copy video PID drop — null-packet
                          * stuffing instead of segment abort/restart.
                          *
                          * Root cause context: disk I/O stalls on the
                          * segment write path (sda at 98.4% utilisation,
                          * 472ms avg await) cause the kernel UDP receive
                          * buffer to fill and drop entire 1316-byte UDP
                          * datagrams (7 TS packets each — confirmed:
                          * 158M receive buffer errors in netstat -su,
                          * CC gaps consistently 6-7 packets per event).
                          *
                          * What TSDuck does (confirmed by comparing the
                          * same source piped through TSDuck→ffmpeg vs
                          * udp_hls direct): on detecting a CC gap, it
                          * writes (gap) null TS packets (PID=0x1FFF) to
                          * fill the missing slots, maintaining stream
                          * timing/continuity. The H.264 decoder conceals
                          * the missing frame(s) — at most a brief
                          * macroblock for ~2ms — then resumes normally.
                          * NO segment restart, NO IDR wait, NO stall.
                          *
                          * What the old approach did (seg_abort then
                          * recovering=1): deleted the current segment
                          * from the playlist, then blocked all video
                          * writes until the next IDR (0-2s). ExoPlayer
                          * saw a gap in the playlist → stall/freeze for
                          * the full IDR wait period, far worse than the
                          * decoder-level concealment TSDuck achieves.
                          *
                          * The null stuffing cap (16 max = one full CC
                          * cycle) prevents runaway stuffing if the gap
                          * value is somehow corrupted or wraps; normal
                          * disk-stall drops produce gaps of 6-7.       */
                         if(!c->opts.video_transcode&&pid==c->vid_pid
                            &&c->seg_fd>=0&&c->started){
                             int gap=(int)((cc-(uint8_t)((prev+1)&0xF))&0xF);
                             if(gap<1)gap=1;
                             if(gap>16)gap=16;
                             /* PCR-interpolated stuffing: write gap packets
                              * with smoothly advancing PCR values so the
                              * player's clock reference stays continuous.
                              * This is what TSDuck does and why
                              * TSDuck→ffmpeg plays clean while plain null
                              * stuffing (PID=0x1FFF, no PCR) still causes
                              * player drops — the PCR discontinuity at the
                              * gap boundary is what ExoPlayer actually
                              * reacts to, not the missing video data.
                              *
                              * Packet rate at 5.5Mbps ≈ 3657 pkts/sec.
                              * Each missing packet = 1/3657s ≈ 273µs.
                              * In 27MHz PCR ticks: 273µs × 27000000 ≈ 7371
                              * per packet. We write gap packets spaced by
                              * this increment so the PCR advances linearly
                              * across the stuffed gap at the correct rate.
                              * If last_pcr27 is unknown (-1), fall back to
                              * plain null packets.                         */
                             if(c->last_pcr27>=0&&c->pcr_pid){
                                 /* 27MHz PCR ticks per TS packet at nominal
                                  * bitrate. At 5.5Mbps: one 188-byte packet
                                  * takes 188*8/5500000 seconds ≈ 273µs.
                                  * In 27MHz ticks: 273µs × 27000000 ≈ 7371.
                                  * Fine-grained accuracy isn't critical here
                                  * — what matters is that PCR advances
                                  * monotonically and at roughly the right
                                  * rate across the gap so the player's clock
                                  * reference doesn't see a discontinuity.  */
                                 const int64_t TICKS_PER_PKT=7371;
                                 uint8_t pcr_pkt[188];
                                 for(int _n=0;_n<gap;_n++){
                                     int64_t pcr=c->last_pcr27+
                                                 (int64_t)(_n+1)*TICKS_PER_PKT;
                                     ts_write_pcr_pkt(pcr_pkt,c->pcr_pid,pcr);
                                     pkt_write(c,pcr_pkt);}
                                 c->last_pcr27+=
                                     (int64_t)(gap+1)*TICKS_PER_PKT;
                             }else{
                                 for(int _n=0;_n<gap;_n++)
                                     pkt_write(c,(const uint8_t*)NULL_TS_PKT);}
                             /* No seg_abort, no recovering=1 —
                              * stream continues uninterrupted.
                              * Continuation packets after the gap
                              * are written as-is: H.264 uses unbounded
                              * PES (length=0) so the decoder scans for
                              * NAL start codes rather than relying on
                              * PES length — a mid-NAL gap produces
                              * macroblocks in the affected region but
                              * the decoder keeps running. Discarding
                              * continuations (tried in V1.0.46) caused
                              * WORSE visible freezes: decoder had no
                              * data and repeated the last good frame
                              * for the entire gap duration, which is
                              * more disruptive than brief macroblocks.*/
                         }else if(c->seg_fd>=0&&!video_recovery_throttled){
                             /* For simple-copy audio PID drops: also use
                              * null stuffing rather than seg_close(). The
                              * audio decoder conceals a brief gap (a few
                              * missing MP2/AAC frames = a few ms of silence)
                              * far more gracefully than a full segment
                              * restart -- which causes a 0.1-1.5s stall
                              * while the player re-syncs. Confirmed from
                              * real logs: 0x0642 audio drops occurring
                              * simultaneously with video drops (same UDP
                              * datagram loss event) were still triggering
                              * seg_close(), producing short segments (0.339s,
                              * 0.113s) that caused player freezes even after
                              * the video-PID stuffing was fixed.
                              * For video_transcode, keep seg_close() since
                              * the encoder needs clean boundaries.        */
                             if(!c->opts.video_transcode){
                                 int gap=(int)((cc-(uint8_t)((prev+1)&0xF))&0xF);
                                 if(gap<1)gap=1;if(gap>16)gap=16;
                                 for(int _n=0;_n<gap;_n++)
                                     pkt_write(c,(const uint8_t*)NULL_TS_PKT);
                             }else{seg_close(c,now);c->seg_timing_ok=0;}}}
                     /* Only set recovering for video_transcode VIDEO-PID
                      * drop — for simple-copy the null stuffing above
                      * keeps the stream live without needing recovery. */
                     if(pid==c->vid_pid&&!video_recovery_throttled
                        &&c->opts.video_transcode){c->recovering=1;c->recover_wait_ok=0;}
                     /* A CC drop means we lost packets — any AU currently
                      * being accumulated is now corrupt/incomplete, and
                      * the decoder's reference picture buffer may
                      * reference frames that were never fully received.
                      * Force a full re-sync: discard whatever partial AU
                      * is in flight and require a fresh SPS+IDR (which
                      * also flushes the decoder) before decoding resumes.
                      * See vidxc_push's dec_ready gate for the flush.
                      * This fires independently of whatever other PID
                      * may already be mid-recovery — see rationale above
                      * for why that independence matters.               */
                     if(c->opts.video_transcode&&c->vidxc&&pid==c->vid_pid&&!video_recovery_throttled){
                         c->vidxc->dec_ready=0;
                         c->vidxc->force_idr=1;
                         au_reset(&c->vidxc->au);
                         c->last_recovery_trigger=*now;
                         c->last_recovery_trigger_valid=1;
                     }
                     /* Same rationale as above for video, applied to
                      * audio: a CC drop means real elapsed time has
                      * diverged from "samples decoded so far" — the
                      * encoder's PTS clock must re-lock onto the next
                      * packet's real PTS rather than keep extrapolating
                      * from its original (now stale) anchor point.     */
                     int asi=aud_slot_for_pid(c,pid);
                     if(asi>=0){
                         c->aud[asi].audxc.pts_ok=0;
                         c->aud[asi].audxc.pusi_seen=0;
                         c->aud[asi].audxc.last_seen_source_pts_valid=0;
                         c->aud[asi].audxc.latency_measured=0;
                         c->aud[asi].audxc.samples_since_anchor=0;
                         c->aud[asi].audxc.anchor_pts=0;
                         c->aud[asi].audxc.cumulative_drift_correction=0;
                     }}
             }
         }
         c->cc_last[pid&0x1FFF]=cc;}}

    /* ── STATE 0: wait for first IDR ──────────────────────────
     * Cache SPS/PPS as they arrive. Start segment on first IDR.
     * IDR detection works on BOTH PUSI and continuation packets
     * because some sources split SPS+IDR across two TS packets. */
    if(!c->started){
        if(c->pmt_ok&&c->vid_pid&&pid==c->vid_pid&&ts_pusi(p)){
            /* video_transcode=1: start on any PUSI — encoder will open segment
             * on first IDR output (see vidxc block below).
             * simple-copy / audio_transcode: require source IDR.              */
            int can_start = c->opts.video_transcode ? 1 :
                (c->opts.simple_cut ? (nal_type(p)==5||nal_type(p)==20)
                                     : has_keyframe(p));
            /* Hybrid startup fallback — see MAX_START_WAIT_SECS doc.
             * Prefer a clean IDR/has_keyframe() start, but don't let a
             * source that never satisfies that check sit here forever:
             * track the first candidate video PUSI seen while waiting,
             * and once MAX_START_WAIT_SECS has passed with no qualifying
             * keyframe, start on the current PUSI regardless — restores
             * v23's guarantee that a channel always eventually starts.
             * video_transcode is unaffected: can_start is already
             * unconditionally 1 there, so this branch never engages.   */
            if(!can_start&&!c->opts.video_transcode){
                if(!c->start_wait_ok){
                    c->start_wait_start=*now;c->start_wait_ok=1;
                }else if(wall_el(&c->start_wait_start,now)>=(double)MAX_START_WAIT_SECS){
                    can_start=1;
                    fprintf(stderr,"[%s] WARNING: no IDR/keyframe seen within "
                            "%ds of first video PUSI — starting on plain PUSI "
                            "(v23-compatible fallback)\n",c->name,MAX_START_WAIT_SECS);
                }
            }
            if(can_start){
                c->started=1;c->start_wait_ok=0;
                {Evt e={EVT_START,time(NULL),0,0,0,c->vid_pid,0,0};evpush(&c->evring,&e);}
                /* For video_transcode: seg_open happens on first encoder IDR.
                 * For simple-copy: open segment now and write first IDR.      */
                if(!c->opts.video_transcode){
                    seg_open(c,now);pkt_write(c,p);}
            }
        }
        if(!c->started)return;
        if(!c->opts.video_transcode)return;}

    /* ── STATE 1: CC recovery ───────────────────────────────── */
    /* This gate exists ONLY to hold back video output until a clean
     * random-access point arrives after a recovery — it must never
     * affect any other PID. Confirmed as a real, serious requirement:
     * audio (and anything else) must never be blocked by video
     * recovery state, even briefly. The old structure's final `else
     * return` fired for ANY non-video-PUSI packet while recovering was
     * true, including every audio packet, silently discarding them
     * until video recovered — this is now scoped so packets on any
     * PID other than the video PID skip this block entirely and fall
     * straight through to their normal processing, completely
     * unaffected by c->recovering's value.                           */
    if(c->recovering&&pid==c->vid_pid){
        if(ts_pusi(p)){
            if(c->opts.video_transcode){
                /* Don't open a segment here — that races against
                 * vidxc_push's own I-slice gate and our encoder's GOP
                 * cadence (see rationale above). Just clear the
                 * recovering flag; vidxc_push will silently discard
                 * non-I-slice AUs until a genuine I-slice arrives
                 * (dec_ready gate), and the existing got_keyframe/
                 * enc_started logic below opens the segment exactly
                 * the same way it does for the very first segment —
                 * only once OUR OWN encoder actually emits a fresh
                 * IDR with freshly re-sent SPS/PPS.                  */
                c->recovering=0;c->recover_wait_ok=0;
            }else{
                int can_recover=c->opts.simple_cut?
                    (nal_type(p)==5||nal_type(p)==20):has_keyframe(p);
                int recovered_via_fallback=0;
                /* Same hybrid fallback as STATE 0 startup above — see
                 * MAX_START_WAIT_SECS doc. Without it, a source that
                 * never produces a qualifying IDR/keyframe would stay
                 * stuck in c->recovering=1 permanently after a CC drop,
                 * silently discarding every subsequent video packet via
                 * the `else return` below — NO_SIGNAL/STALLED for the
                 * rest of the session, one CC drop away on every
                 * affected channel.                                    */
                if(!can_recover){
                    if(!c->recover_wait_ok){
                        c->recover_wait_start=*now;c->recover_wait_ok=1;
                    }else if(wall_el(&c->recover_wait_start,now)>=(double)MAX_START_WAIT_SECS){
                        can_recover=1;recovered_via_fallback=1;
                        fprintf(stderr,"[%s] WARNING: no IDR/keyframe seen within "
                                "%ds of CC-drop recovery — recovering on plain "
                                "PUSI (v23-compatible fallback)\n",c->name,MAX_START_WAIT_SECS);
                    }
                }
                if(can_recover){
                    c->recovering=0;c->recover_wait_ok=0;
                    /* recovered_via_fallback means THIS packet was never
                     * verified as a keyframe — same reasoning as the
                     * open_gop/forced-cut suppress_replay fix in the
                     * segment-cut block: skip the stale cached SPS/PPS
                     * replay rather than risk reference-buffer corruption
                     * ahead of an unverified packet.                    */
                    seg_open_ex(c,now,recovered_via_fallback);
                    pkt_write(c,p);
                }else return;
            }
        }else return; /* continuation packet on the video PID during
            recovery — still correctly dropped: an incomplete/stale AU
            must not be accumulated. Only ever applies to the video
            PID now, never to anything else.                          */
        if(!c->opts.video_transcode)return;}

    /* ── video_transcode: decode+re-encode ─────────────────── */
    if(c->opts.video_transcode&&c->vidxc&&c->vidxc->active&&c->vid_pid&&pid==c->vid_pid){
        if(!c->seg_timing_ok){c->seg_start=*now;c->seg_timing_ok=1;}
        c->vidxc->got_keyframe=0;
        int n=vidxc_push(c->vidxc,p,c->vid_pid);
        if(c->vidxc->got_keyframe){
            if(!c->vidxc->enc_started){
                /* First IDR from encoder: now open the first segment.
                 * This ensures segment 0 always starts with IDR+audio. */
                c->vidxc->enc_started=1;
                if(c->seg_fd<0) seg_open(c,now); /* open if not already */
                printf("[vidxc] first IDR — segment opened, streaming starts\n");
            } else if(c->seg_fd<0){
                /* No segment currently open (e.g. immediately after CC
                 * recovery closed the previous one) — open one now
                 * regardless of elapsed time. This is the point where
                 * our own encoder has just produced a genuine fresh
                 * IDR with freshly re-sent SPS/PPS, making it exactly
                 * the right and only safe moment to start a new
                 * segment after a gap.                                */
                seg_open(c,now);
            } else if(c->seg_timing_ok){
                /* Subsequent IDR: cut segment if time elapsed.
                 * Small tolerance (10% of SEGMENT_SECS) absorbs normal
                 * keyframe-arrival jitter around the exact threshold —
                 * see rationale above this block for why this matters:
                 * missing an early arrival means waiting a FULL extra
                 * GOP for the next one (e.g. 2.0s threshold missed by
                 * 0.06s at elapsed=1.94s produced a 3.98s segment,
                 * confirmed via direct timing trace), which is a much
                 * worse outcome than a segment slightly under target.  */
                double el=wall_el(&c->seg_start,now);
                double tol=(double)SEGMENT_SECS*0.10;
                if(el>=(double)SEGMENT_SECS-tol){
                    seg_close(c,now);seg_open(c,now);}
            }
        }
        for(int i=0;i<n;i++)pkt_write(c,c->vidxc->out+i*TS_SZ);
        return;}

    /* ── Segment cut (simple-copy / audio_transcode) ────────── */
    if(c->vid_pid&&pid==c->vid_pid){
        if(ts_pusi(p)){
            int nt=nal_type(p);
            if(nt==7){/* SPS: reset cache */
                c->spspps_n=0;
                memcpy(c->spspps[0],p,TS_SZ);c->spspps_n=1;}
            else if(nt==8&&c->spspps_n>0&&c->spspps_n<8){
                memcpy(c->spspps[c->spspps_n],p,TS_SZ);c->spspps_n++;}
            /* Independent check, NOT an else-if: a packet classified as
             * nt==7 (SPS) by nal_type()'s own priority ranking can still
             * carry a real keyframe slice in the same payload (confirmed
             * on this source — SPS and the keyframe slice travel
             * together). Making this mutually exclusive with the SPS
             * branch above was the actual regression: the cut check
             * never ran at all for exactly the packets it needed to.   */
            int is_cut_point;
            if(c->open_gop){
                /* v23-compatible open-GOP mode (see OPEN_GOP_MISS_THRESHOLD
                 * doc): this source has already proven it never produces a
                 * detectable IDR/keyframe cut point, so stop looking and
                 * treat every video PUSI as a valid cut candidate, exactly
                 * like v23's idr_off. Restores smooth ~SEGMENT_SECS
                 * segments instead of paying the full MAX_SEG_WAIT_SECS
                 * toll on every single segment forever.                   */
                is_cut_point=1;
            } else if(c->opts.simple_cut){
                /* v23-compatible: raw NAL type only, no Exp-Golomb.
                 * Cuts ONLY on a genuine IDR (type 5/20) — sources using
                 * nal_type=1 for functional keyframes (the reason
                 * has_keyframe exists) will rarely/never satisfy this,
                 * same real limitation v23 itself has on those sources.*/
                is_cut_point=(nt==5||nt==20);
            } else {
                is_cut_point=(nt==5||nt==20||has_keyframe(p));
            }
            if(is_cut_point){/* IDR, or (default mode only) a real
                I-slice carried on a non-IDR NAL type (see has_keyframe
                doc) — either way, a genuine random-access point, safe
                to cut.                                                 */
                if(c->seg_timing_ok){
                    double el=wall_el(&c->seg_start,now);
                    double tol=(double)SEGMENT_SECS*0.10;
                    if(el>=(double)SEGMENT_SECS-tol){
                        /* If THIS packet just updated the SPS cache
                         * (nt==7), it already carries fresh SPS/PPS
                         * itself — seg_open()'s normal cache-replay
                         * would write this identical packet, then the
                         * pkt_write(c,p) below writes it AGAIN right
                         * after, producing a genuine duplicate
                         * SPS+PPS+slice sequence at the segment
                         * boundary (confirmed via direct byte
                         * inspection: this was the actual cause of the
                         * "illegal short term buffer state" decode
                         * errors, not a timing problem). Suppress the
                         * cache replay for this one seg_open() call —
                         * the immediately-following pkt_write supplies
                         * the fresh parameter sets instead.            */
                        /* suppress_replay must fire whenever THIS exact
                         * packet already carries fresh SPS/PPS itself,
                         * not just when nal_type()'s own priority-ranked
                         * return value happens to be 7. nal_type()
                         * returns IDR(5/20) immediately if present,
                         * even when SPS/PPS appear earlier in the SAME
                         * packet payload (confirmed real case: NALs
                         * [9,7,8,6,5] — AUD, SPS, PPS, SEI, IDR all in
                         * one packet, nal_type() correctly reports 5).
                         * The original fix only checked nt==7, missing
                         * this exact scenario — confirmed directly via
                         * real failing segments showing the predicted
                         * duplicate-SPS+PPS corruption pattern
                         * ("reference picture missing"/"mmco: unref
                         * short failure" cascades) at segment
                         * boundaries with zero CC drops anywhere
                         * nearby, ruling out packet loss as the cause.*/
                        int pkt_has_sps=0;
                        {int po=ts_poff(p);if(po>=0){
                            const uint8_t*pp=p+po;int pl=TS_SZ-po;
                            if(ts_pusi(p)&&pl>=9&&pp[0]==0&&pp[1]==0&&pp[2]==1){
                                int sk=9+pp[8];if(sk<pl){pp+=sk;pl-=sk;}}
                            for(int k=0;k+2<pl;k++){
                                int n2=-1;
                                if(pp[k]==0&&pp[k+1]==0&&pp[k+2]==1)n2=k+3;
                                else if(k+3<pl&&pp[k]==0&&pp[k+1]==0&&pp[k+2]==0&&pp[k+3]==1)n2=k+4;
                                if(n2<0||n2>=pl)continue;
                                if((pp[n2]&0x1F)==7){pkt_has_sps=1;break;}
                                k=n2;}}}
                        /* open_gop cuts land on an ARBITRARY video PUSI —
                         * not a verified I-slice, just whatever packet
                         * happened to be next once timing allowed a cut
                         * (see is_cut_point/open_gop above). Forcing the
                         * normal cached-SPS/PPS replay here injects fresh
                         * parameter sets immediately ahead of a packet
                         * that is frequently a P/B-slice with no IDR at
                         * all — many decoders read "new SPS/PPS just
                         * arrived" as a strong hint that a clean random-
                         * access point follows, then get a slice that
                         * references pictures from the PREVIOUS segment
                         * (which no longer exists in the same file), the
                         * exact recipe for the reference-buffer
                         * corruption / visible freezes and macroblocking
                         * confirmed on real playback once open_gop
                         * engaged. v23's own idr_off cut path writes
                         * ONLY PAT+PMT at these blind cuts — no SPS/PPS —
                         * relying on the decoder's existing (already-
                         * active) parameter sets and treating the cut as
                         * a pure container/file boundary, not a fresh
                         * random-access point. Match that exactly here:
                         * always suppress the cache replay in open_gop
                         * mode, regardless of what nt/pkt_has_sps say.   */
                        int suppress_replay=c->open_gop||(nt==7)||pkt_has_sps;
                        if(!c->open_gop)c->seg_wait_misses=0; /* real cut
                            succeeded — this source is still producing
                            detectable keyframes, so the consecutive-miss
                            streak toward open-GOP resets.               */
                        seg_close(c,now);seg_open_ex(c,now,suppress_replay);
                        if(c->opts.idr_rewrite&&!c->open_gop&&nt!=5&&nt!=20){
                            /* Only valid for a has_keyframe()-verified
                             * I-slice cut point (see has_keyframe/
                             * rewrite_slice_to_idr docs). Never apply
                             * this to an open_gop blind cut: that packet
                             * is not confirmed to be an I-slice at all —
                             * it could be a P/B-slice, and mislabeling
                             * its NAL type as an IDR (5) tells the
                             * decoder to reset its reference-picture
                             * buffer to empty right before feeding it a
                             * slice that still needs those references —
                             * guaranteed corruption, not a fix.         */
                            /* This is exactly the cut-point packet, and
                             * it got here via has_keyframe()'s Exp-
                             * Golomb check (nt is 1, not a real IDR) —
                             * see rewrite_slice_to_idr()/idr_rewrite doc
                             * for full rationale. Mutate a local copy;
                             * p itself is const and may be read again
                             * by other code paths after this returns.  */
                            uint8_t tmp[TS_SZ];memcpy(tmp,p,TS_SZ);
                            rewrite_slice_to_idr(tmp);
                            pkt_write(c,tmp);return;}
                        pkt_write(c,p);return;}}}
            /* CRITICAL: this fallback must run on EVERY PUSI packet,
             * not just ones identified as a cut-point candidate above —
             * confirmed via precise comparison against v23 (which has
             * this as an unconditional statement at this same scope,
             * not nested inside its IDR-found branch). If seg_timing_ok
             * is ever false when a NON-keyframe PUSI packet arrives,
             * leaving this nested inside is_cut_point meant it would
             * never get initialized from this path at all — seg_start
             * would stay stale, corrupting every subsequent
             * wall_el(&c->seg_start,now) duration calculation and the
             * tolerance check above, with no trace in CC-error logging
             * since this has nothing to do with CC detection. This was
             * a real, previously-undetected regression versus v23's
             * behavior, found by direct line-by-line comparison after
             * ruling out has_keyframe/simple_cut as the cause via real
             * production testing.                                      */
            if(!c->seg_timing_ok){c->seg_start=*now;c->seg_timing_ok=1;}
            /* Hard timeout fallback: if we've been waiting for the
             * regular has_keyframe()/IDR cut point for far longer than
             * normal (see MAX_SEG_WAIT_SECS doc), force a cut on THIS
             * PUSI packet even though it isn't one — confirmed real
             * cases of this exact gap (two genuine 14.2s segments, no
             * CC drop involved) left a real player frozen for the
             * whole gap with zero log signal otherwise. A forced cut
             * here isn't materially worse than a normal one on this
             * source: every segment boundary already lacks a true IDR
             * (confirmed via direct inspection of real captured
             * segments: zero NAL type 5 anywhere in 8 consecutive
             * segments, only type-1 slices that satisfy has_keyframe()'s
             * Exp-Golomb check) and already shows reference-buffer
             * corruption (mmco/reference-picture errors) on every
             * decode regardless — bounding the wait time turns an
             * unbounded freeze into, at worst, the same kind of brief
             * concealable glitch every other segment already has.      */
            else if(c->seg_timing_ok&&!is_cut_point){
                double waited=wall_el(&c->seg_start,now);
                if(waited>=(double)MAX_SEG_WAIT_SECS){
                    c->seg_wait_misses++;
                    static time_t _last_warn=0;
                    time_t now_t=time(NULL);
                    if(now_t!=_last_warn){
                        fprintf(stderr,"[%s] WARNING: forced segment cut after "
                                "%.1fs waiting for next keyframe (no CC drop "
                                "involved) — source's keyframe cadence is "
                                "running longer than normal\n",c->name,waited);
                        _last_warn=now_t;}
                    /* v23-compatible permanent fallback — see
                     * OPEN_GOP_MISS_THRESHOLD doc. A single forced cut can
                     * be a genuine one-off cadence anomaly (the original
                     * reason MAX_SEG_WAIT_SECS exists) and shouldn't give
                     * up on real IDR detection permanently. But repeated
                     * back-to-back forced cuts mean this source simply
                     * never produces a point has_keyframe()/simple_cut can
                     * detect — paying the full MAX_SEG_WAIT_SECS toll on
                     * EVERY segment forever is exactly the jerky,
                     * oversized-segment behavior v23 doesn't have on the
                     * same source, so switch permanently to open-GOP
                     * (cut on any PUSI) instead, same as v23's idr_off.   */
                    if(!c->open_gop&&c->seg_wait_misses>=OPEN_GOP_MISS_THRESHOLD){
                        c->open_gop=1;
                        fprintf(stderr,"[%s] open-GOP stream: no detectable "
                                "IDR/keyframe after %d consecutive forced cuts "
                                "— cutting on any PUSI from now on "
                                "(v23-compatible)\n",c->name,c->seg_wait_misses);
                    }
                    /* Same reasoning as the open_gop suppress_replay fix
                     * above: THIS packet was never verified as a keyframe
                     * either (that's precisely why we're here — is_cut_point
                     * was false) — replaying cached SPS/PPS ahead of it
                     * risks the same reference-buffer corruption. Use
                     * seg_open_ex(...,1) instead of plain seg_open().     */
                    seg_close(c,now);seg_open_ex(c,now,1);pkt_write(c,p);return;
                }
            }}}

    /* ── audio_transcode: intercept audio PES ─────────────────
     * Use PUSI guard: discard continuation before first PUSI.
     * Dispatches by which active audio slot (if any) this PID
     * belongs to — multiple tracks can be active simultaneously,
     * each with its own independent transcoder/passthrough state.   */
    if(c->opts.audio_transcode){
        int si=aud_slot_for_pid(c,pid);
        if(si>=0){
            AudSlot*s=&c->aud[si];
            if(!s->audxc.active)goto passthrough;
            int oo=ts_poff(p);if(oo<0)goto passthrough;
            const uint8_t*pes=p+oo;int plen=TS_SZ-oo;
            if(ts_pusi(p)){
                if(plen<9||pes[0]!=0||pes[1]!=0||pes[2]!=1)goto passthrough;
                uint8_t pf=(pes[7]>>6)&3;
                if(pf&&plen>=14){
                    int64_t src_pts=(int64_t)(
                        ((uint64_t)(pes[9]&0x0E)<<29)|((uint64_t)pes[10]<<22)|
                        ((uint64_t)(pes[11]&0xFE)<<14)|((uint64_t)pes[12]<<7)|
                        ((uint64_t)(pes[13]&0xFE)>>1));
                    /* Always update last_seen_source_pts — needed by
                     * the latency measurement in audxc_feed_pcm, which
                     * may not run until several PUSI packets after the
                     * anchor below was first captured.                */
                    s->audxc.last_seen_source_pts=src_pts;
                    s->audxc.last_seen_source_pts_valid=1;
                    if(!s->audxc.pts_ok){
                        s->audxc.pts=src_pts;s->audxc.pts_ok=1;
                        s->audxc.anchor_pts=src_pts;
                        s->audxc.cumulative_drift_correction=0;}
                    /* Ongoing clock-drift correction — restored (see
                     * anchor_pts/cumulative_drift_correction field docs
                     * for the full A/B evidence against a working
                     * ffmpeg reference on a real ~2hr run). Runs on
                     * every PUSI, driven by real source PES arrivals
                     * rather than AAC frame cadence, gently nudging
                     * `pts` toward what the source's live clock
                     * actually says right now — instead of letting
                     * pts_inc's fixed-nominal-48kHz assumption free-run
                     * uncorrected for the life of the channel. Gated on
                     * latency_measured so this never fights with the
                     * one-time startup correction in audxc_feed_pcm —
                     * that one runs first and folds its result into
                     * cumulative_drift_correction too, so this check's
                     * "already applied" baseline is accurate from the
                     * very first ongoing correction onward.            */
                    if(s->audxc.pts_ok&&s->audxc.latency_measured&&
                       s->audxc.samples_since_anchor>0){
                        int64_t expected_src_pts=s->audxc.anchor_pts+
                            (s->audxc.samples_since_anchor*90000LL/48000)+
                            s->audxc.cumulative_drift_correction;
                        int64_t gap=s->audxc.last_seen_source_pts-expected_src_pts;
                        /* Sanity bound intentionally wide (+/-1 hour) —
                         * only rejects genuinely nonsensical values (a
                         * real 2^33-tick PTS wraparound, decode
                         * garbage), never limits how much real
                         * accumulated drift can be corrected. A tight
                         * bound here previously, silently, refused to
                         * correct exactly the channels that needed it
                         * most the moment their gap grew past it.       */
                        if(gap>-324000000LL&&gap<324000000LL&&gap!=0){
                            /* Only a small fraction of the observed gap
                             * per check — this runs many times per
                             * second, so it still converges quickly
                             * while staying well under anything
                             * perceptible as a single-step jump.
                             * Magnitude capped to pts_inc/16: large
                             * enough to converge in reasonable time,
                             * small enough that no realistic number of
                             * nudges landing between two AAC frame
                             * emissions can add up to a full frame
                             * step — which is what caused a real
                             * regression (full audio drop, from a
                             * non-monotonic output PTS) the first time
                             * this was built without the cap.          */
                            int64_t nudge=gap/16;
                            int64_t max_nudge=s->audxc.pts_inc>0?
                                s->audxc.pts_inc/16:1;
                            if(max_nudge<1)max_nudge=1;
                            if(nudge>max_nudge)nudge=max_nudge;
                            if(nudge<-max_nudge)nudge=-max_nudge;
                            if(nudge==0)nudge=(gap>0)?1:-1;
                            s->audxc.pts+=nudge;
                            s->audxc.cumulative_drift_correction+=nudge;
                        }
                    }}
                int skip=9+pes[8];
                if(skip>=plen)goto passthrough;
                pes+=skip;plen-=skip;
                s->audxc.pusi_seen=1;}
            else{
                if(!s->audxc.pusi_seen)goto passthrough;}
            int nout=audxc_push(&s->audxc,&c->aud_scratch,pes,plen,s->out_pid,&s->cc);
            static int _adone[MAX_CHANNELS][MAX_AUD_TRACKS];
            int chidx=(int)(c-g_ch);
            if(nout>0&&chidx>=0&&chidx<MAX_CHANNELS&&!_adone[chidx][si]){
                printf("[%s] audxc track%d: first AAC output: %d bytes (%d TS pkts)\n",
                       c->name,si+1,nout,nout/TS_SZ);
                _adone[chidx][si]=1;}
            if(nout>0)s->write_loop_entries++;
            for(int off=0;off+TS_SZ<=nout;off+=TS_SZ){
                pkt_write(c,s->audxc.out+off);
                s->write_loop_packets++;}
            return;}}
    passthrough:;

    /* ── zero-CPU patches (pts_reset, fix_interlace) ────────── */
    if(c->opts.pts_reset||c->opts.fix_interlace){
        uint8_t tmp[TS_SZ];memcpy(tmp,p,TS_SZ);
        if(c->opts.pts_reset&&(pid==c->vid_pid||aud_slot_for_pid(c,pid)>=0)&&ts_pusi(tmp))
            pes_patch_pts(tmp,&c->pts_base,&c->pts_base_ok);
        if(c->opts.fix_interlace&&pid==c->vid_pid&&ts_pusi(tmp))
            sps_patch_interlace(tmp);
        pkt_write(c,tmp);return;}

    pkt_write(c,p);}

/* ════════════════════════════════════════════════════════════════
   NIC RECEIVE THREAD
   ════════════════════════════════════════════════════════════════ */
typedef struct{Ch**chs;int nch;char iface[64];}NicGroup;

/* Interface resolution cache, keyed by multicast group.
 *
 * V1.0.42 — added after a confirmed real production problem: channel
 * reconnects on signal loss are implemented as a full process restart
 * (wrapper script/systemd), which means sock_open() — and therefore
 * the full probe_iface_for_group() join/poll/drop sequence — re-runs
 * LIVE every time ANY channel reconnects, while every other channel
 * is actively streaming on the same shared NICs. With 300 channels
 * and routine transient signal loss, this was happening frequently,
 * not as a rare edge case. Confirmed mechanism: IP_ADD_MEMBERSHIP and
 * IP_DROP_MEMBERSHIP both issue real IGMP reports onto the physical
 * network segment — repeated join/leave churn during a reconnect can
 * disrupt IGMP-snooping switches' forwarding state for a multicast
 * group, affecting OTHER channels sharing that same group (confirmed
 * earlier in this file's own history: CNBC_TV_18 and
 * KALAIGNAR_ISAI_ARUVI share group 231.1.1.3), which is consistent
 * with the reported symptom — continuous CC (continuity counter)
 * errors / ExoPlayer drops and A/V sync loss, persisting for hours,
 * not just around a one-time startup window.
 *
 * Fix: remember the resolved device for each multicast group on first
 * successful resolution, and reuse it instantly on every subsequent
 * channel start/restart for that same group — skipping the live probe
 * (and therefore all its IGMP join/leave churn) entirely on
 * reconnects. Only falls through to a fresh live probe if there's no
 * cached entry yet, OR if the cached device no longer exists on this
 * box at all (e.g. NIC renamed/removed since the cache was written —
 * checked directly via getifaddrs() before ever trusting a cache hit,
 * so a stale entry can never strand a channel). Deliberately does NOT
 * re-validate "is this device still actually receiving traffic" on
 * every reconnect — that would reintroduce exactly the join/leave
 * churn this fix removes. A cached device that's stopped carrying the
 * right traffic (NIC wiring changed, not just renamed/removed) is a
 * separate, rarer failure mode this doesn't address — if that's ever
 * suspected, clear /tmp/udp_hls_iface_cache.txt to force fresh probing
 * for every group again.
 *
 * Concurrency: many channel processes can start together (e.g. after
 * a box reboot) and read/write this file at the same time. Reads use
 * a shared flock, writes use an exclusive flock around a full read-
 * modify-write, so concurrent writes from different channels never
 * corrupt each other's entries — verified directly with 50 concurrent
 * writer processes, zero malformed lines, zero duplicate entries,
 * zero lost writes.                                                   */
#define IFACE_CACHE_PATH "/tmp/udp_hls_iface_cache.txt"
#define IFACE_CACHE_MAX_LINES 4096

static int iface_cache_lookup(const char*mcast_ip,char*out_dev,size_t out_len){
    FILE*f=fopen(IFACE_CACHE_PATH,"r");
    if(!f)return 0;
    if(flock(fileno(f),LOCK_SH)<0){fclose(f);return 0;}
    char line[256];
    int found=0;
    while(fgets(line,sizeof line,f)){
        char ip[64]={0},dev[64]={0};
        if(sscanf(line,"%63s %63s",ip,dev)!=2)continue; /* skip any
            malformed/partial line rather than fail the whole lookup —
            tolerates a line from a write that was interrupted before
            this fix's flock-based protection existed, or any future
            manual edits to the cache file.                            */
        if(strcmp(ip,mcast_ip)==0){
            strncpy(out_dev,dev,out_len-1);out_dev[out_len-1]=0;
            found=1;break;}
    }
    flock(fileno(f),LOCK_UN);
    fclose(f);
    if(!found)return 0;
    /* Never trust a cached device blindly — confirm it still exists on
     * this box right now before using it, so a stale entry (NIC
     * renamed/removed since the cache was written) safely falls
     * through to a fresh live probe instead of stranding the channel
     * on a device that no longer exists.                              */
    struct ifaddrs*ifs=NULL;getifaddrs(&ifs);
    int still_exists=0;
    for(struct ifaddrs*i=ifs;i;i=i->ifa_next)
        if(i->ifa_name&&strcmp(i->ifa_name,out_dev)==0){still_exists=1;break;}
    if(ifs)freeifaddrs(ifs);
    return still_exists;}

static void iface_cache_store(const char*mcast_ip,const char*dev){
    int fd=open(IFACE_CACHE_PATH,O_RDWR|O_CREAT,0644);
    if(fd<0)return;
    if(flock(fd,LOCK_EX)<0){close(fd);return;}
    FILE*f=fdopen(fd,"r+");
    if(!f){close(fd);return;}
    static char lines[IFACE_CACHE_MAX_LINES][256];
    int nlines=0,replaced=0;
    char line[256];
    rewind(f);
    while(nlines<IFACE_CACHE_MAX_LINES&&fgets(line,sizeof line,f)){
        char ip[64]={0},olddev[64]={0};
        if(sscanf(line,"%63s %63s",ip,olddev)==2&&strcmp(ip,mcast_ip)==0){
            snprintf(lines[nlines],sizeof lines[nlines],"%s %s\n",mcast_ip,dev);
            replaced=1;
        }else{
            strncpy(lines[nlines],line,sizeof lines[nlines]-1);
            lines[nlines][sizeof lines[nlines]-1]=0;}
        nlines++;}
    if(!replaced&&nlines<IFACE_CACHE_MAX_LINES){
        snprintf(lines[nlines],sizeof lines[nlines],"%s %s\n",mcast_ip,dev);
        nlines++;}
    rewind(f);
    if(ftruncate(fileno(f),0)==0)
        for(int i=0;i<nlines;i++)fputs(lines[i],f);
    fflush(f);
    flock(fileno(f),LOCK_UN);
    fclose(f);} /* closes fd too */

/* Listen-and-confirm interface detection — ported from a Java
 * implementation the user already had and wanted matched exactly: for
 * each candidate local interface (up, non-loopback, multicast-capable
 * — the same three checks Java's NetworkInterface exposes via
 * isUp()/isLoopback()/supportsMulticast(), here read directly from
 * ifa_flags via getifaddrs()), actually join the given multicast group
 * on THAT interface and wait up to `timeout_ms` for one real packet.
 * Returns 1 and fills out_ip on the first interface that receives a
 * packet, 0 if none did within the per-interface timeout.
 *
 * This is the STRONGEST guarantee of the three resolution strategies —
 * it confirms real traffic is actually flowing on that interface right
 * now, rather than just asking the routing table what it WOULD use
 * (resolve_iface_for_group(), below) or guessing the first non-
 * loopback interface system-wide (auto_iface()). The real cost: up to
 * (timeout_ms × number of candidate interfaces) in the worst case if
 * the source isn't currently flowing on any of them. Confirmed
 * acceptable with the user: this runs once per channel at startup
 * provisioning time, not repeatedly on a hot path, so the extra wall-
 * clock cost here is an explicit, accepted tradeoff for the stronger
 * guarantee — this is why it's tried FIRST in sock_open() below, ahead
 * of the near-instant routing-table method, rather than only as a
 * fallback.                                                            */
static int probe_iface_for_group(const char*mcast_ip,int port,int timeout_ms,
                                  char*out_ip,size_t out_ip_len){
    struct ifaddrs*ifs=NULL;
    if(getifaddrs(&ifs)<0)return 0;
    int found=0;
    for(struct ifaddrs*i=ifs;i&&!found;i=i->ifa_next){
        if(!i->ifa_addr||i->ifa_addr->sa_family!=AF_INET)continue;
        if(!(i->ifa_flags&IFF_UP)){
            fprintf(stderr,"[probe] %s: skipped (not UP)\n",i->ifa_name);continue;}
        if(i->ifa_flags&IFF_LOOPBACK){
            fprintf(stderr,"[probe] %s: skipped (loopback)\n",i->ifa_name);continue;}
        if(!(i->ifa_flags&IFF_MULTICAST)){
            fprintf(stderr,"[probe] %s: skipped (no multicast flag)\n",i->ifa_name);continue;}
        struct sockaddr_in*sa=(struct sockaddr_in*)i->ifa_addr;
        char dbg_ip[INET_ADDRSTRLEN];inet_ntop(AF_INET,&sa->sin_addr,dbg_ip,sizeof dbg_ip);
        struct timespec t0,t1;clock_gettime(CLOCK_MONOTONIC,&t0);
        fprintf(stderr,"[probe] %s (%s): trying, timeout=%dms...\n",i->ifa_name,dbg_ip,timeout_ms);

        int fd=socket(AF_INET,SOCK_DGRAM,0);
        if(fd<0)continue;
        int one=1;
        setsockopt(fd,SOL_SOCKET,SO_REUSEADDR,&one,sizeof one);
        /* V1.0.41 FIX — SO_BINDTODEVICE, confirmed necessary by direct
         * test against real production hardware: when another udp_hls
         * channel sharing the SAME multicast group/port is already
         * running (a real, routine condition for this deployment — two
         * different channel names can record the same upstream feed),
         * its already-active socket's IP_ADD_MEMBERSHIP join means the
         * kernel delivers a copy of every packet to ANY OTHER socket
         * also bound to that (group,port), regardless of what THAT
         * socket's own imr_interface was set to. Confirmed directly:
         * a probe socket joined with imr_interface set to the box's
         * default-route NIC (which has no physical connection to the
         * actual feed) still received a real, valid TS packet
         * (got_n=1316, first_byte=0x47) in 1ms, purely because another
         * channel's process already had a correctly-joined socket on
         * the same group/port — imr_interface alone does NOT prevent
         * this. SO_BINDTODEVICE is different: it restricts the SOCKET
         * ITSELF to a specific physical device for both send and
         * receive, independent of multicast group membership state on
         * other sockets/devices. Confirmed by direct test in both
         * directions: bound to the real device, traffic is received
         * normally (no regression); bound to a real-but-disconnected
         * device, a concurrently-active correct socket's traffic on
         * the right device does NOT leak through (request times out,
         * as it should) — this is the one mechanism of the three tried
         * so far that's actually immune to the cross-talk this
         * deployment routinely exhibits. Failure here (e.g. requires
         * CAP_NET_RAW — expected to be available since this already
         * runs as root in production) is non-fatal: just skip this
         * candidate and move on, same fail-safe convention as every
         * other step in this loop.                                    */
        if(setsockopt(fd,SOL_SOCKET,SO_BINDTODEVICE,i->ifa_name,strlen(i->ifa_name))<0){
            fprintf(stderr,"[probe] %s: SO_BINDTODEVICE failed (%s), skipping\n",
                    i->ifa_name,strerror(errno));
            close(fd);continue;}
        /* Bind to the MULTICAST GROUP address itself, the same
         * strategy sock_open() already uses for the real receive
         * socket (see `sa.sin_addr.s_addr=grp;` there). This was
         * tested two other ways first and both were wrong, confirmed
         * by direct send/receive test rather than assumed:
         *   - INADDR_ANY (the original draft): works for RECEIVING,
         *     but doesn't scope reception to a specific interface at
         *     all on its own — see the SO_BINDTODEVICE rationale above
         *     for why imr_interface alone isn't enough either.
         *   - A specific unicast interface address (the first attempted
         *     fix): confirmed by direct local send/receive test to
         *     NEVER receive multicast traffic at all — Linux UDP socket
         *     demux matches on the packet's actual destination address,
         *     which for multicast is always the GROUP address, never a
         *     unicast address, so a socket bound to a unicast address
         *     simply never matches an incoming multicast datagram
         *     regardless of IP_ADD_MEMBERSHIP/imr_interface.
         * Binding to the group address, confirmed by the same kind of
         * direct test, correctly receives multicast traffic for that
         * group.                                                       */
        struct sockaddr_in bindaddr={0};
        bindaddr.sin_family=AF_INET;
        bindaddr.sin_port=htons((uint16_t)port);
        if(inet_pton(AF_INET,mcast_ip,&bindaddr.sin_addr)!=1){close(fd);continue;}
        if(bind(fd,(struct sockaddr*)&bindaddr,sizeof bindaddr)<0){close(fd);continue;}

        struct ip_mreq mr={0};
        mr.imr_multiaddr=bindaddr.sin_addr;
        mr.imr_interface=sa->sin_addr;
        if(setsockopt(fd,IPPROTO_IP,IP_ADD_MEMBERSHIP,&mr,sizeof mr)<0){close(fd);continue;}

        struct pollfd pfd={.fd=fd,.events=POLLIN};
        int pr=poll(&pfd,1,timeout_ms);
        int got_ts=0;ssize_t got_n=-1;uint8_t first_byte=0;
        if(pr>0&&(pfd.revents&POLLIN)){
            uint8_t buf[1500];
            ssize_t n=recv(fd,buf,sizeof buf,0);
            got_n=n;if(n>0)first_byte=buf[0];
            /* Require the payload to actually look like MPEG-TS (sync
             * byte 0x47), the same check used everywhere else in this
             * file to validate a packet before trusting it (see
             * pkt_process's own `if(p[0]!=0x47)return;`). Any arriving
             * datagram is not proof this is the right feed on its own
             * — this guards against unrelated traffic incidentally
             * sharing this group/port being misread as a match.        */
            if(n>0&&buf[0]==0x47){
                found=1;got_ts=1;
                inet_ntop(AF_INET,&sa->sin_addr,out_ip,(socklen_t)out_ip_len);
            }
        }
        clock_gettime(CLOCK_MONOTONIC,&t1);
        double elapsed_ms=(t1.tv_sec-t0.tv_sec)*1000.0+(t1.tv_nsec-t0.tv_nsec)/1e6;
        fprintf(stderr,"[probe] %s (%s): poll_result=%d got_n=%zd first_byte=0x%02X "
                "match=%s elapsed=%.0fms\n",
                i->ifa_name,dbg_ip,pr,got_n,first_byte,got_ts?"YES":"no",elapsed_ms);
        /* Drop membership before closing — this fd is just a probe,
         * never the channel's real receive socket (that's opened
         * fresh, separately, in sock_open()), so leave no trace of
         * group membership registered against this address.          */
        setsockopt(fd,IPPROTO_IP,IP_DROP_MEMBERSHIP,&mr,sizeof mr);
        close(fd);
    }
    if(ifs)freeifaddrs(ifs);
    fprintf(stderr,"[probe] FINAL: %s\n",found?"matched (see iface above)":"no interface matched, falling back");
    return found;}

/* Resolve which local interface IP the kernel would actually use to
 * reach a given multicast group, without sending any packet and
 * without forking a subprocess.
 *
 * V1.0.37: added to remove the need to hand-map each of 300 channels'
 * multicast groups to the correct NIC IP when the box has multiple
 * physical/bonded interfaces each carrying different feeds (confirmed
 * real operational pain point — auto_iface() below only ever returns
 * ONE interface system-wide, which silently breaks any channel whose
 * group isn't reachable on that particular NIC; mr.imr_interface joins
 * the group on the wrong interface, the join itself doesn't fail, and
 * the channel just never receives any packets — see sock_open()'s
 * SO_REUSEPORT removal comment for the same class of "join succeeds,
 * traffic silently never arrives" failure mode this avoids repeating).
 *
 * Mechanism: connect() on an unconnected UDP socket performs a real
 * kernel routing-table lookup and binds a local address accordingly —
 * no SYN, no data, nothing is sent on the wire for a UDP socket. This
 * is the same lookup `ip route get <mcast>` reports (confirmed
 * directly: a real multicast group on this exact production box
 * resolved via a normal routing-table entry — "dev eth0 src X" with no
 * "via" gateway hop — meaning no special PIM/igmpproxy multicast
 * routing is configured here, just a plain unicast-style lookup, which
 * is exactly what connect() resolves internally). getsockname()
 * afterward reports which local address the kernel chose, which tells
 * us which interface.
 *
 * Guarded to only ever attempt this for a genuine multicast address
 * (224.0.0.0/4 — same test already used as `mc` in sock_open() below).
 * Confirmed by direct test: without this guard, a degenerate/malformed
 * config value like "0.0.0.0" silently resolves to the loopback
 * interface (127.0.0.1) via this same connect()/getsockname() trick —
 * which would be a real, silent misconfiguration if this were ever fed
 * a bad config line, joining a "multicast group" on `lo` and receiving
 * nothing, exactly the kind of clean-looking-but-empty failure this
 * feature exists to avoid introducing elsewhere.
 *
 * Returns 1 on success (out_ip filled), 0 on any failure — caller must
 * always have a fallback path (the existing iface/auto_iface() chain)
 * and must never block channel startup on this.                       */
static int resolve_iface_for_group(const char*mcast_ip,int port,char*out_ip,size_t out_len){
    struct in_addr grp_addr;
    if(inet_pton(AF_INET,mcast_ip,&grp_addr)!=1)return 0;
    if((ntohl(grp_addr.s_addr)>>28)!=0xE)return 0; /* not 224.0.0.0/4 */
    int fd=socket(AF_INET,SOCK_DGRAM,0);
    if(fd<0)return 0;
    struct sockaddr_in dst={0};
    dst.sin_family=AF_INET;
    dst.sin_port=htons((uint16_t)(port>0?port:1));
    dst.sin_addr=grp_addr;
    if(connect(fd,(struct sockaddr*)&dst,sizeof dst)<0){close(fd);return 0;}
    struct sockaddr_in local={0};socklen_t slen=sizeof local;
    if(getsockname(fd,(struct sockaddr*)&local,&slen)<0){close(fd);return 0;}
    close(fd);
    if(local.sin_addr.s_addr==0)return 0; /* unresolved/INADDR_ANY */
    if(!inet_ntop(AF_INET,&local.sin_addr,out_ip,(socklen_t)out_len))return 0;
    return 1;}

static void auto_iface(char*out,size_t len){
    struct ifaddrs*l=NULL;getifaddrs(&l);
    for(struct ifaddrs*i=l;i;i=i->ifa_next){
        if(!i->ifa_addr||i->ifa_addr->sa_family!=AF_INET)continue;
        struct sockaddr_in*sa=(struct sockaddr_in*)i->ifa_addr;
        if((ntohl(sa->sin_addr.s_addr)>>28)==0)continue;
        if((ntohl(sa->sin_addr.s_addr)>>24)==127)continue;
        inet_ntop(AF_INET,&sa->sin_addr,out,(socklen_t)len);break;}
    if(l)freeifaddrs(l);}

static int sock_open(Ch*c){
    uint32_t grp=inet_addr(c->mcast);
    int mc=((ntohl(grp)>>28)==0xE);
    char iface[64]={0};
    int auto_resolved=0;
    if(c->iface[0])strncpy(iface,c->iface,63);
    else if(mc){
        /* No iface given (config has "-" or omits the field) — three
         * resolution strategies are tried in order, strongest
         * guarantee first:
         *
         * 1. probe_iface_for_group() — actually join the group on
         *    each candidate interface and wait for a real packet
         *    (ported from the user's own Java implementation, by
         *    request). Confirms genuine live traffic, not just a
         *    routing-table opinion — the strongest guarantee of the
         *    three. Cost: up to ~1.5s per candidate interface in the
         *    worst case if the source isn't currently flowing
         *    anywhere; confirmed acceptable with the user since this
         *    runs once per channel at startup, not on a repeated hot
         *    path.
         * 2. resolve_iface_for_group() — near-instant routing-table
         *    query (no actual packet wait), tried only if the probe
         *    above found nothing (e.g. source briefly silent at the
         *    exact moment of startup).
         * 3. auto_iface() — single-interface, system-wide guess, last
         *    resort only if both of the above fail.
         *
         * Never blocks or fails channel startup over this — any
         * resolution failure just falls through to the next strategy,
         * and total failure still leaves a usable (if likely wrong)
         * interface value rather than refusing to start.              */
        if(probe_iface_for_group(c->mcast,c->port,1500,iface,sizeof iface))
            auto_resolved=1;
        else if(resolve_iface_for_group(c->mcast,c->port,iface,sizeof iface))
            auto_resolved=1;
        else
            auto_iface(iface,sizeof iface);
        /* Record what was actually used back onto the channel struct —
         * c->iface is otherwise only ever the raw config value (empty
         * here), and without this, startup logging/diagnostics have no
         * way to show which interface a "-"/auto channel actually
         * ended up joining on. Harmless: nothing reads c->iface before
         * this point in a channel's lifecycle (no reload/restart path
         * re-enters sock_open() with stale expectations about it).    */
        if(iface[0])strncpy(c->iface,iface,63);
    }
    int fd=socket(AF_INET,SOCK_DGRAM,IPPROTO_UDP);
    if(fd<0){perror("socket");return -1;}
    int one=1,zero=0;
    setsockopt(fd,SOL_SOCKET,SO_REUSEADDR,&one,sizeof one);
    /* SO_REUSEPORT intentionally NOT set.
     *
     * Confirmed real bug this caused: two channels sharing the same
     * UDP port (different multicast groups, same NIC) each had
     * SO_REUSEPORT set. When the primary bind() to the multicast group
     * address (below) falls through to its INADDR_ANY fallback -- a
     * real, observed path on production servers, not hypothetical --
     * the resulting socket becomes eligible to receive ANY UDP traffic
     * on that port regardless of destination multicast group, and the
     * kernel's SO_REUSEPORT delivery selection among multiple sockets
     * bound to the same (address, port) does NOT consult each socket's
     * own IP_ADD_MEMBERSHIP group membership when choosing which
     * socket gets a given datagram. Net effect, confirmed directly via
     * decrypted production segments: a second channel's process
     * silently received and decoded the FIRST channel's video/audio
     * (same PAT/PMT/PCR PIDs found inside the second channel's own
     * output segments), while genuinely looking for its own (different)
     * audio PID on every packet -- which never arrived, since it was
     * never actually receiving its own multicast stream. Zero error
     * anywhere: aud_slot_for_pid() correctly returned no match for
     * every packet it WAS given, so audio was silently never
     * transcoded, with completely clean stats (cc_tot stable,
     * idle=0.0s, segs incrementing on schedule) since the wrong-
     * channel's packets were still real, valid, continuously-arriving
     * MPEG-TS the whole time.
     *
     * This design runs exactly one socket per channel with no
     * intentional load-sharing across processes/sockets -- the actual
     * use case SO_REUSEPORT exists for -- so removing it costs
     * nothing functionally, and prevents this exact cross-talk class
     * of bug even if two channels end up sharing a port again.        */
    int rb=RCVBUF;
    if(setsockopt(fd,SOL_SOCKET,SO_RCVBUFFORCE,&rb,sizeof rb)!=0)
        setsockopt(fd,SOL_SOCKET,SO_RCVBUF,&rb,sizeof rb);
    fcntl(fd,F_SETFL,fcntl(fd,F_GETFL,0)|O_NONBLOCK);
    struct sockaddr_in sa={0};
    sa.sin_family=AF_INET;sa.sin_port=htons((uint16_t)c->port);
    if(mc){
        sa.sin_addr.s_addr=grp;
        if(bind(fd,(struct sockaddr*)&sa,sizeof sa)<0){
            /* Diagnostic: this fallback is exactly the path that made
             * the cross-talk bug above possible in the first place --
             * a socket bound to INADDR_ANY:port instead of the
             * specific multicast group address has no kernel-level
             * destination-address filtering of its own, relying
             * entirely on IP_ADD_MEMBERSHIP (an accept-onto-host
             * filter, not a per-socket delivery-routing filter) for
             * correctness. Logged so this is visible per-channel at
             * startup instead of silently happening.                  */
            fprintf(stderr,"[%s] WARNING: bind() to multicast group %s:%d "
                    "failed (%s) -- falling back to INADDR_ANY:%d. This "
                    "socket will rely entirely on IP_ADD_MEMBERSHIP for "
                    "correct packet delivery; if another channel shares "
                    "this port, verify it is not also on this fallback "
                    "path.\n",c->name,c->mcast,c->port,strerror(errno),c->port);
            sa.sin_addr.s_addr=htonl(INADDR_ANY);
            bind(fd,(struct sockaddr*)&sa,sizeof sa);}
        struct ip_mreq mr={0};
        mr.imr_multiaddr.s_addr=grp;
        mr.imr_interface.s_addr=iface[0]?inet_addr(iface):htonl(INADDR_ANY);
        if(setsockopt(fd,IPPROTO_IP,IP_ADD_MEMBERSHIP,&mr,sizeof mr)<0){
            perror("IP_ADD_MEMBERSHIP");close(fd);return -1;}
        setsockopt(fd,IPPROTO_IP,IP_MULTICAST_ALL,&zero,sizeof zero);
    }else{sa.sin_addr.s_addr=htonl(INADDR_ANY);bind(fd,(struct sockaddr*)&sa,sizeof sa);}
    return fd;}

static void mkdirp(const char*p){
    char t[512];snprintf(t,512,"%s",p);
    for(char*s=t+1;*s;s++)if(*s=='/'){*s=0;mkdir(t,0755);*s='/';}
    mkdir(t,0755);}

static void*nic_thread(void*arg){
    NicGroup*g=(NicGroup*)arg;int nch=g->nch;
    struct pollfd*pfds=calloc(nch,sizeof*pfds);
    av_log_set_level(AV_LOG_ERROR); /* suppress decoder warnings */
    av_log_set_callback(corruption_log_callback); /* observe real
        reference-frame corruption signals - see corruption_log_callback
        and corrupt_errors_since_check docs for full rationale          */

    for(int i=0;i<nch;i++){
        Ch*c=g->chs[i];
        mkdirp(c->dir);
        if(!c->seg_seq)c->seg_seq=c->start_seq;
        c->seg_fd=-1;c->pat_ok=0;c->pmt_ok=0;
        c->seg_timing_ok=0;c->started=0;c->recovering=0;c->start_wait_ok=0;c->recover_wait_ok=0;c->open_gop=0;c->seg_wait_misses=0;
        c->last_pcr27=-1;c->cc_drop_skip=0;
        c->wb_n=0;c->aes_on=0;
        memset(c->cc_last,0xFF,sizeof c->cc_last);c->cc_errors=0;
        c->pts_base_ok=0;c->pts_base=0;
        c->fps_probe_have_last=0;c->fps_probe_n=0;c->fps_probe_done=0;
        c->fps_probe_mbs_only=-1;
        c->fps_probe_ambiguous=0;c->fps_probe_extra_pkts=0;
        c->fps_probe_pending_fpn=0;c->fps_probe_pending_fpd=0;
        c->fps_probe_pending_rep=0;
        c->target_dur=0;c->target_dur_fixed=0;c->spspps_n=0;
        memset(c->aud,0,sizeof c->aud);c->n_aud_active=0;c->n_aud_dropped=0;
        c->vidxc=NULL; /* heap-allocated lazily on first video_transcode=1
            init in pkt_process -- guaranteed NULL here since nic_thread
            runs this block exactly once per channel per process
            lifetime (no reload/restart path re-enters it).           */
        c->snap_pkts=0;c->snap_bytes=0;c->snap_cc=0;
        clock_gettime(CLOCK_MONOTONIC,&c->snap_time);
        memset(&c->evring,0,sizeof c->evring);
        clock_gettime(CLOCK_MONOTONIC,&c->last_pkt);

        /* AES key */
        if(c->keyfile[0]&&strcmp(c->keyfile,"-")!=0){
            FILE*kf=fopen(c->keyfile,"rb");
            if(kf){uint8_t key[16];size_t n=fread(key,1,16,kf);fclose(kf);
                if(n==16){uint8_t iv[16]={0};int ok=1;
                    if(c->iv_hex[0]&&strcmp(c->iv_hex,"-")!=0){
                        if(strlen(c->iv_hex)!=32)ok=0;
                        else for(int j=0;j<16;j++){unsigned v=0;
                            if(sscanf(c->iv_hex+j*2,"%02x",&v)!=1){ok=0;break;}
                            iv[j]=(uint8_t)v;}}
                    if(ok){aes_init(&c->aes,key,iv);c->aes_on=1;
                           printf("[%s] AES ready\n",c->name);}}}}

        c->fd=sock_open(c);
        if(c->fd<0){pfds[i].fd=-1;pfds[i].events=0;continue;}
        pfds[i].fd=c->fd;pfds[i].events=POLLIN;

        const char*mode=(c->opts.audio_transcode||c->opts.video_transcode)?
                        "in-process":"simple-copy";
        printf("[%s] joined %s:%d iface=%s  mode=%s\n",
               c->name,c->mcast,c->port,c->iface[0]?c->iface:"auto",mode);}

    printf("[NIC:%s] watching %d channels\n",
           g->iface[0]?g->iface:"auto",nch);fflush(stdout);

    static uint8_t buf[UDP_MAX];
    while(!g_stop){
        int ready=poll(pfds,nch,10);
        if(ready<0){if(errno==EINTR)continue;break;}
        struct timespec now;clock_gettime(CLOCK_MONOTONIC,&now);
        for(int i=0;i<nch&&!g_stop;i++){
            Ch*c=g->chs[i];
            if(pfds[i].fd<0)continue;
            /* stale detection */
            if(c->started&&c->seg_fd>=0){
                double age=wall_el(&c->last_pkt,&now);
                if(age>=(double)(SEGMENT_SECS*2)){
                    Evt e={EVT_STALE,time(NULL),0,(uint32_t)(age*1000),0,0,0,0};
                    evpush(&c->evring,&e);
                    seg_close(c,&now);c->started=0;c->recovering=0;c->seg_timing_ok=0;c->start_wait_ok=0;c->recover_wait_ok=0;c->open_gop=0;c->seg_wait_misses=0;
                    /* Reset transcode decoder state too, not just
                     * segmenter state — confirmed real gap: this branch
                     * previously only reset c->started/recovering/
                     * seg_timing_ok, which is sufficient for simple-copy
                     * (no decoder state exists there at all) but leaves
                     * video_transcode's h264 decoder and audio_transcode's
                     * mpg123 decoder completely untouched across a real
                     * multi-second signal outage. Feeding fresh, post-gap
                     * packets into decoders that still think they're
                     * mid-stream (stale reference-picture buffer for
                     * h264, stale internal sync state for mpg123) is
                     * exactly the kind of thing that produces corrupted
                     * decode output or silently-broken output — confirmed
                     * real-world symptom on this exact channel: a STALE
                     * event with audio_transcode active left mpg123's
                     * internal state stale after CLEAN START, producing
                     * no audio afterward (and, most likely, the "video
                     * drops" reported alongside it — a player stalling
                     * on a broken/silent audio track inside an otherwise
                     * fine HLS segment looks like video drops from the
                     * viewer's side even when the video track itself is
                     * untouched). Mirrors the EXACT reset already used
                     * for a routine CC drop (see pkt_process's video/
                     * audio CC handling) — a multi-second STALE gap is a
                     * strictly bigger discontinuity than a routine CC
                     * drop, so applying the same-or-stronger reset here
                     * is the safe direction, never weaker.              */
                    if(c->opts.video_transcode&&c->vidxc){
                        c->vidxc->dec_ready=0;
                        c->vidxc->force_idr=1;
                        au_reset(&c->vidxc->au);
                    }
                    for(int ai=0;ai<c->n_aud_active;ai++){
                        AudXc*a=&c->aud[ai].audxc;
                        if(!a->active)continue;
                        /* Full mpg123 resync (close+reopen), not just the
                         * lighter pts/anchor reset used for a routine CC
                         * drop — a multi-second outage is a big enough
                         * discontinuity that any bytes still sitting in
                         * mpg123's internal buffer are almost certainly
                         * stale/torn, same reasoning as the MPG123_ERR
                         * resync path this mirrors (see that block's own
                         * doc for why closing beats trying to salvage).  */
                        mpg123_close(a->mh);
                        if(mpg123_open_feed(a->mh)!=MPG123_OK){
                            fprintf(stderr,"[%s] mpg123_open_feed FAILED on "
                                    "STALE recovery — disabling this audio "
                                    "track\n",c->name);
                            a->active=0;continue;
                        }
                        a->pts_ok=0;a->pusi_seen=0;
                        a->last_seen_source_pts_valid=0;
                        a->latency_measured=0;a->samples_since_anchor=0;
                        a->anchor_pts=0;a->cumulative_drift_correction=0;
                    }
                }}
            if(!(pfds[i].revents&POLLIN))continue;
            for(int lim=0;lim<MAX_PER_CHAN&&!g_stop;lim++){
                ssize_t n=recv(c->fd,buf,sizeof buf,0);
                if(n<=0){if(errno==EAGAIN||errno==EWOULDBLOCK)break;continue;}
                c->dgrams++;clock_gettime(CLOCK_MONOTONIC,&c->last_pkt);
                if(c->dgrams==1){
                    printf("[%s] first dgram len=%d\n",c->name,(int)n);
                    if(buf[0]!=0x47){
                        /* RTP detect */
                        if(n>=13&&(buf[0]>>6)==2){
                            int h=12+(buf[0]&0xF)*4;
                            if((buf[0]>>4)&1&&n>h+4)h+=4+((buf[h+2]<<8)|buf[h+3])*4;
                            if(h<n&&buf[h]==0x47){
                                printf("[%s] RTP+%d\n",c->name,h);
                                /* store rtp offset in dgrams field temporarily */
                                /* Actually just scan from offset */
                                const uint8_t*ts=buf+h;
                                int np=(int)(n-h)/TS_SZ;
                                for(int j=0;j<np;j++,ts+=TS_SZ)
                                    if(ts[0]==0x47)pkt_process(c,ts,&now);
                                continue;}}
                        printf("[%s] unknown format byte=0x%02X\n",c->name,buf[0]);
                        continue;}
                    else printf("[%s] plain TS\n",c->name);}
                /* plain TS */
                int np=(int)n/TS_SZ;
                const uint8_t*ts=buf;
                for(int j=0;j<np;j++,ts+=TS_SZ)
                    if(ts[0]==0x47)pkt_process(c,ts,&now);}}}

    /* shutdown */
    for(int i=0;i<nch;i++){
        Ch*c=g->chs[i];
        if(c->seg_fd>=0){
            struct timespec now;clock_gettime(CLOCK_MONOTONIC,&now);
            if(c->aes_on)aes_flush(&c->aes,c->seg_fd,1);else wb_flush(c);
            close(c->seg_fd);printf("[%s] final\n",c->name);}
        for(int k=0;k<c->n_aud_active;k++)audxc_free(&c->aud[k].audxc);
        if(c->vidxc){vidxc_free(c->vidxc);free(c->vidxc);c->vidxc=NULL;}
        if(c->fd>=0)close(c->fd);}
    free(pfds);return NULL;}

/* ════════════════════════════════════════════════════════════════
   OPTS PARSER / CONFIG LOADER
   ════════════════════════════════════════════════════════════════ */
static void parse_opts(ChOpts*o,const char*toks[],int ntok){
    memset(o,0,sizeof*o);
    strncpy(o->audio_coder,"fast",sizeof(o->audio_coder)-1); /* real default —
        previously "fast" only appeared in a log-message fallback that
        never reached audxc_init; the actual struct field stayed empty
        unless audio_coder= was explicitly passed in config, silently
        falling back to libavcodec's own default (twoloop) instead.   */
    for(int i=0;i<ntok;i++){
        const char*t=toks[i];
        if     (!strncmp(t,"audio_transcode=",16))o->audio_transcode=atoi(t+16);
        else if(!strncmp(t,"video_transcode=",16))o->video_transcode=atoi(t+16);
        else if(!strncmp(t,"fix_mp2=",8))         o->fix_mp2=atoi(t+8);
        else if(!strncmp(t,"fix_interlace=",14))  o->fix_interlace=atoi(t+14);
        else if(!strncmp(t,"pts_reset=",10))       o->pts_reset=atoi(t+10);
        else if(!strncmp(t,"video_crf=",10))       o->video_crf=atof(t+10);
        else if(!strncmp(t,"audio_bitrate=",14))   o->audio_bitrate=atoi(t+14);
        else if(!strncmp(t,"audio_map=",10))       o->audio_map=atoi(t+10);
        else if(!strncmp(t,"simple_cut=",11))       o->simple_cut=atoi(t+11);
        else if(!strncmp(t,"idr_rewrite=",12))      o->idr_rewrite=atoi(t+12);
        else if(!strncmp(t,"audio_coder=",12)){
            strncpy(o->audio_coder,t+12,sizeof(o->audio_coder)-1);
            o->audio_coder[sizeof(o->audio_coder)-1]='\0';
        }}}

static int conf_load(const char*path){
    FILE*f=fopen(path,"r");if(!f){perror(path);return 0;}
    int n=0;char line[2048];
    while(fgets(line,sizeof line,f)&&n<MAX_CHANNELS){
        char*nl=strchr(line,'\n');if(nl)*nl=0;
        char*hsh=strchr(line,'#');if(hsh)*hsh=0;
        int l=(int)strlen(line);
        while(l>0&&(line[l-1]==' '||line[l-1]=='\t'))line[--l]=0;
        if(!line[0])continue;
        Ch*c=&g_ch[n];memset(c,0,sizeof*c);
        unsigned long long seq=0;char opt_buf[1024]={0};
        int r=sscanf(line,
            "%63s %d %255s %63s %63s %255s %511s %32s %llu %1023[^\n]",
            c->mcast,&c->port,c->dir,c->name,
            c->iface,c->keyfile,c->keyuri,c->iv_hex,&seq,opt_buf);
        if(r<4)continue;
        c->start_seq=(uint64_t)seq;
        if(c->iface[0]  &&!strcmp(c->iface,  "-"))c->iface[0]=0;
        if(c->keyfile[0]&&!strcmp(c->keyfile,"-"))c->keyfile[0]=0;
        if(c->keyuri[0] &&!strcmp(c->keyuri, "-"))c->keyuri[0]=0;
        if(c->iv_hex[0] &&!strcmp(c->iv_hex, "-"))c->iv_hex[0]=0;
        if(opt_buf[0]){
            const char*toks[32];int ntok=0;char*sv=NULL;
            char*tok=strtok_r(opt_buf," \t",&sv);
            while(tok&&ntok<32){toks[ntok++]=tok;tok=strtok_r(NULL," \t",&sv);}
            parse_opts(&c->opts,toks,ntok);}
        printf("[conf] %s:%d -> %s/%s  %s%s%s\n",
               c->mcast,c->port,c->dir,c->name,
               c->opts.audio_transcode?"audio_transcode ":"",
               c->opts.video_transcode?"video_transcode ":"",
               (!c->opts.audio_transcode&&!c->opts.video_transcode)?"simple-copy":"");
        n++;}
    fclose(f);return n;}

/* ════════════════════════════════════════════════════════════════
   STATS THREAD
   ════════════════════════════════════════════════════════════════ */
static void dump_channel_state(Ch*c){
    printf("\n[DUMP][%s] ── diagnostic state dump ────────────────────\n",c->name);
    printf("  pat_ok=%d pmt_ok=%d pmt_pid=0x%04X vid_pid=0x%04X pcr_pid=0x%04X "
           "vid_stream_type=0x%02X\n",
           c->pat_ok,c->pmt_ok,c->pmt_pid,c->vid_pid,c->pcr_pid,c->vid_stream_type);
    printf("  started=%d recovering=%d seg_fd=%d seg_seq=%llu segs=%llu\n",
           c->started,c->recovering,c->seg_fd,
           (unsigned long long)c->seg_seq,(unsigned long long)c->segs);
    printf("  n_aud_active=%d n_aud_dropped=%d\n",c->n_aud_active,c->n_aud_dropped);
    for(int i=0;i<c->n_aud_active;i++){
        AudSlot*s=&c->aud[i];
        printf("    aud[%d] src_pid=0x%04X out_pid=0x%04X stream_type=0x%02X cc=0x%X "
               "audxc.active=%d audxc.src_codec=%d audxc.pts_ok=%d audxc.pusi_seen=%d "
               "audxc.pts=%lld audxc.pts_inc=%lld audxc.push_calls=%llu "
               "audxc.push_calls_with_output=%llu write_loop_entries=%llu "
               "write_loop_packets=%llu\n",
               i,s->src_pid,s->out_pid,s->stream_type,s->cc,
               s->audxc.active,(int)s->audxc.src_codec,s->audxc.pts_ok,s->audxc.pusi_seen,
               (long long)s->audxc.pts,(long long)s->audxc.pts_inc,
               (unsigned long long)s->audxc.push_calls,
               (unsigned long long)s->audxc.push_calls_with_output,
               (unsigned long long)s->write_loop_entries,
               (unsigned long long)s->write_loop_packets);
    }
    for(int i=0;i<c->n_aud_dropped;i++)
        printf("    aud_dropped[%d] pid=0x%04X\n",i,c->aud_dropped[i]);
    printf("  cc_last[vid_pid&0x1FFF]=0x%02X",
           c->vid_pid?c->cc_last[c->vid_pid&0x1FFF]:0xFF);
    for(int i=0;i<c->n_aud_active;i++)
        printf("  cc_last[aud[%d]_pid&0x1FFF]=0x%02X",
               i,c->cc_last[c->aud[i].src_pid&0x1FFF]);
    printf("\n  cc_errors_total=%u\n",c->cc_errors);
    if(c->opts.video_transcode){
        if(c->vidxc){
            printf("  vidxc.active=%d vidxc.dec_ready=%d vidxc.force_idr=%d "
                   "vidxc.stall_calls=%d vidxc.got_keyframe=%d vidxc.enc_started=%d\n",
                   c->vidxc->active,c->vidxc->dec_ready,c->vidxc->force_idr,
                   c->vidxc->stall_calls,c->vidxc->got_keyframe,c->vidxc->enc_started);
        } else {
            printf("  vidxc=NULL (not yet allocated -- video_transcode=1 set but "
                   "no PMT-known video PUSI packet seen yet)\n");
        }
    }
    printf("[DUMP][%s] ──────────────────────────────────────────────\n\n",c->name);
}
static void*stats_run(void*a){(void)a;
    int tick=0;
    while(!g_stop){
        for(int s=0;s<10&&!g_stop;s++){
            sleep(1);
            if(g_dump_requested){
                g_dump_requested=0;
                printf("\n[DUMP] SIGUSR1 received — dumping state for %d channel(s)\n",g_nch);
                for(int i=0;i<g_nch;i++)dump_channel_state(&g_ch[i]);
                fflush(stdout);
            }
        }
        if(g_stop)break;
        tick++;
        struct timespec now;clock_gettime(CLOCK_MONOTONIC,&now);
        char ts[9];ts_now(ts);
        uint64_t tsegs=0,tpkts=0;int stalled=0,nosig=0;
        printf("\n[%s] ── stats ──────────────────────────────────\n",ts);
        for(int i=0;i<g_nch;i++){
            Ch*c=&g_ch[i];
            evdrain(&c->evring,c->name);
            double idle=wall_el(&c->last_pkt,&now);
            if(idle<0)idle=0; /* harmless cross-thread race between the
                NIC receive thread (writes c->last_pkt on every packet)
                and this stats thread (reads it once per tick) can
                briefly produce a tiny negative value at high packet
                rates — clamp for a sane display, no functional impact
                either way since this is purely a printed value here. */
            double dt=wall_el(&c->snap_time,&now);if(dt<0.001)dt=0.001;
            uint64_t dpkts=c->pkts-c->snap_pkts;
            uint64_t dbytes=c->snap_bytes;
            uint64_t dcc=c->cc_errors-c->snap_cc;
            double pps=(double)dpkts/dt;
            double kbps=(double)dbytes*8.0/dt/1000.0;
            c->snap_pkts=c->pkts;c->snap_bytes=0;c->snap_cc=c->cc_errors;c->snap_time=now;
            tsegs+=c->segs;tpkts+=c->pkts;
            const char*state;
            if(!c->started){state="NO_SIGNAL";nosig++;}
            else if(idle>SEGMENT_SECS*3){state="STALLED";stalled++;}
            else state="OK";
            const char*mode=(c->opts.audio_transcode||c->opts.video_transcode)?
                            "in-process":"simple-copy";
            printf("  [%-14s] %s  segs=%-5llu  %6.1fpkt/s  %7.1fkbps"
                   "  idle=%4.1fs  cc_new=%-3llu  cc_tot=%-5u  mode=%s\n",
                   c->name,state,(unsigned long long)c->segs,
                   pps,kbps,idle,(unsigned long long)dcc,c->cc_errors,mode);
            if(!c->started)
                printf("    !! NO_SIGNAL: PAT=%s PMT=%s vid_pid=0x%04X\n",
                       c->pat_ok?"ok":"miss",c->pmt_ok?"ok":"miss",c->vid_pid);
            if(c->open_gop)
                printf("    ~~ open-GOP stream: cutting on any PUSI\n");
            if(dcc>0)
                printf("    !! CC ERRORS: %llu new (total=%u)\n",
                       (unsigned long long)dcc,c->cc_errors);}
        printf("  ── total: segs=%llu pkts=%llu  stalled=%d nosignal=%d / %d\n",
               (unsigned long long)tsegs,(unsigned long long)tpkts,stalled,nosig,g_nch);
        printf("[%s] ──────────────────────────────────────────\n\n",ts);
        fflush(stdout);}
    return NULL;}

/* ════════════════════════════════════════════════════════════════
   MAIN
   ════════════════════════════════════════════════════════════════ */
int main(int argc,char*argv[]){
    setvbuf(stdout,NULL,_IOLBF,4096); /* line-buffered: reliable log ordering */
    signal(SIGINT,onsig);signal(SIGTERM,onsig);signal(SIGPIPE,SIG_IGN);
    signal(SIGUSR1,onsig_dump);
    /* g_ch is `static Ch g_ch[MAX_CHANNELS]` -- static storage duration
     * globals are GUARANTEED zero-initialized by the C standard before
     * main() ever runs (the BSS segment). A memset(g_ch,0,sizeof g_ch)
     * here would be redundant for correctness, but NOT free: writing
     * zero to every byte of a ~142MB array forces the kernel to
     * actually commit and back every page with real physical memory
     * immediately, rather than leaving untouched pages backed by the
     * shared, copy-on-write zero-page (the normal, cheap state for
     * BSS memory nothing has written to yet).
     *
     * Confirmed directly via /proc/[pid]/smaps_rollup + pmap on a real
     * running instance: a single process running exactly ONE channel
     * showed ~139MB of Private_Dirty anonymous memory, traced via pmap
     * to one large [ anon ] region, and confirmed via objdump on an
     * actual production binary build to be this exact memset call
     * (immediate operand 0x87b5000 = 142,295,040 bytes = sizeof(g_ch)
     * targeting the g_ch symbol, first call in main()).
     *
     * This is safe to omit because every code path that actually
     * POPULATES a channel slot already zeroes that specific slot
     * itself right before writing to it (conf_load's
     * `Ch*c=&g_ch[n];memset(c,0,sizeof*c);` and the direct-CLI mode's
     * equivalent for g_ch[0] below), and every place that ITERATES
     * g_ch[] is bounded by g_nch, never by MAX_CHANNELS -- so
     * whichever slots are never populated are also never read.
     *
     * In a one-process-per-channel deployment (many processes, each
     * configured with just 1 real channel out of the MAX_CHANNELS=512
     * static array), this was the dominant per-process memory cost:
     * ~139MB x N processes. At 300 processes this alone accounted for
     * roughly 40GB of real-world memory growth. DO NOT re-add this
     * memset call -- this exact line has been removed once already
     * and reintroduced by mistake at least once since (copy-paste from
     * an older snapshot of this file); if it's back, it's a
     * regression, not an intentional change.                          */
    if(argc==2&&(!strcmp(argv[1],"--version")||!strcmp(argv[1],"-v")||!strcmp(argv[1],"-V"))){
        print_version();
        return 0;}
    const char*e;
    if((e=getenv("HLS_DELETE_THRESHOLD"))){int t=atoi(e);if(t>=0)g_del=t;}
    if(argc==2){
        g_nch=conf_load(argv[1]);
        if(!g_nch){fprintf(stderr,"no channels\n");return 1;}
    }else if(argc>=5){
        Ch*c=&g_ch[0];memset(c,0,sizeof*c);
        strncpy(c->mcast,argv[1],63);c->port=atoi(argv[2]);
        strncpy(c->dir,argv[3],255);strncpy(c->name,argv[4],63);
        if(argc>5)strncpy(c->iface,  argv[5],63);
        if(argc>6)strncpy(c->keyfile,argv[6],255);
        if(argc>7)strncpy(c->keyuri, argv[7],511);
        if(argc>8)strncpy(c->iv_hex, argv[8],32);
        if(argc>9)c->start_seq=(uint64_t)atoll(argv[9]);
        /* "-" placeholder normalization for the direct-CLI-argument
         * path, matching what the channels.conf line-parsing path
         * already does right after its sscanf() a bit further down.
         * Required for auto interface resolution to activate at all:
         * without this, c->iface literally holds the two-character
         * string "-" (non-empty), so sock_open()'s `if(c->iface[0])`
         * check treats it as a real (but invalid) interface value and
         * skips straight past every resolution strategy below,
         * passing "-" itself into inet_addr()/IP_ADD_MEMBERSHIP and
         * failing with ENODEV.                                        */
        if(c->iface[0]  &&!strcmp(c->iface,  "-"))c->iface[0]=0;
        if(c->keyfile[0]&&!strcmp(c->keyfile,"-"))c->keyfile[0]=0;
        if(c->keyuri[0] &&!strcmp(c->keyuri, "-"))c->keyuri[0]=0;
        if(c->iv_hex[0] &&!strcmp(c->iv_hex, "-"))c->iv_hex[0]=0;
        if(argc>10){
            const char*toks[32];int ntok=0;
            for(int i=10;i<argc&&ntok<32;i++)toks[ntok++]=argv[i];
            parse_opts(&c->opts,toks,ntok);}
        g_nch=1;
    }else{
        print_version();
        fprintf(stderr,
            "UDP->HLS  (all-player: VLC, ExoPlayer, LG, Samsung)\n\n"
            "Usage:\n"
            "  %s channels.conf\n"
            "  %s <mcast> <port> <dir> <name> [iface] [keyfile] [keyuri] [iv] [seq] [opts]\n"
            "  %s --version | -v          show version and exit\n\n"
            "Options:\n"
            "  audio_transcode=1   MP2/MP3->AAC (required for VLC/ExoPlayer)\n"
            "  video_transcode=1   video re-encode (only for MPEG-2 source)\n"
            "  video_crf=23        CRF for video_transcode (default 23)\n"
            "  audio_bitrate=128000 AAC bitrate for audio_transcode (default 128000)\n"
            "  audio_coder=fast    AAC algorithm: twoloop (default, best quality)\n"
            "                      or fast (~2.4x lower CPU, measured)\n"
            "  audio_map=N         select ONLY the Nth audio track (1-indexed);\n"
            "                      default: process ALL audio tracks found\n"
            "  fix_mp2=1           PMT patch only, zero CPU\n"
            "  simple_cut=1        v23-compatible raw-IDR-only segment cut\n"
            "                      (simple-copy/audio_transcode only)\n"
            "  idr_rewrite=1       rewrite cut-point slice NAL type 1->5 (IDR)\n"
            "                      for sources with no real IDRs (default off,\n"
            "                      see source comments before enabling)\n"
            "  fix_interlace=1     SPS progressive patch for LG/Samsung 1080i\n"
            "  pts_reset=1         reset PTS to near-zero\n\n"
            "Build:\n"
            "  gcc -O2 -pthread -o udp_hls udp_hls.c \\\n"
            "      -lssl -lcrypto -lmpg123 -lavcodec -lavutil -lswresample -lm\n\n"
            "Examples:\n"
            "  # H264 + MP2, interlaced 1080i (JAYA TV HD):\n"
            "  %s 236.1.3.78 1026 /hls/jaya jaya eth0 - - - 0  audio_transcode=1 fix_interlace=1\n\n"
            "  # H264 + MP2, progressive:\n"
            "  %s 237.1.1.17 1001 /hls/star star eth0 - - - 0  audio_transcode=1\n\n"
            "  # MPEG-2 + MP2 (old SD channel):\n"
            "  %s 235.1.1.10 1000 /hls/old old eth0 - - - 0  audio_transcode=1 video_transcode=1\n",
            argv[0],argv[0],argv[0],argv[0],argv[0],argv[0]);
        return 1;}

    NicGroup groups[MAX_NICS];int ngroups=0;
    memset(groups,0,sizeof groups);
    for(int i=0;i<MAX_NICS;i++)groups[i].chs=malloc(MAX_CHANNELS*sizeof(Ch*));
    for(int i=0;i<g_nch;i++){
        Ch*c=&g_ch[i];const char*key=c->iface[0]?c->iface:"";
        int gi=-1;
        for(int j=0;j<ngroups;j++)if(!strcmp(groups[j].iface,key)){gi=j;break;}
        if(gi<0){gi=ngroups++;strncpy(groups[gi].iface,key,63);}
        groups[gi].chs[groups[gi].nch++]=c;}
    printf("\n=== ");print_version();printf("    ch=%d  groups=%d  seg=%ds ===\n\n",
           g_nch,ngroups,SEGMENT_SECS);
    pthread_t gtids[MAX_NICS],stid;
    pthread_create(&stid,NULL,stats_run,NULL);
    for(int i=0;i<ngroups;i++)
        pthread_create(&gtids[i],NULL,nic_thread,&groups[i]);
    for(int i=0;i<ngroups;i++)pthread_join(gtids[i],NULL);
    g_stop=1;pthread_join(stid,NULL);
    for(int i=0;i<MAX_NICS;i++)free(groups[i].chs);
    return 0;}
