/*
  UDP -> HLS  —  no-ffmpeg, all-player compatible
  =================================================
  BUILD:
    gcc -O2 -pthread -o udp_hls udp_hls.c \
        -lssl -lcrypto -lmpg123 -lavcodec -lavutil -lswresample -lm
    # Install: apt-get install libopenh264-dev  OR  libx264-dev
    # OpenH264 is preferred; libx264 used as fallback

  channels.conf (one line per channel):
    <mcast> <port> <dir> <name> <iface> <keyfile> <keyuri> <iv> <seq> [opts]

  OPTS:
    audio_transcode=1   MP2/MP3 -> AAC-LC  (~5% CPU, required for ExoPlayer/VLC)
    fix_mp2=1           PMT patch only 0x03/0x04->0x0F, zero CPU
    fix_interlace=1     H264 SPS progressive patch, zero CPU (LG 1080i fix)
    pts_reset=1         reset PTS/DTS to near-zero, zero CPU
    video_transcode=1   H264/MPEG2 -> H264 re-encode (~100% CPU) — only needed
                        when source is MPEG-2; NOT needed for H264 source
    video_crf=23        CRF quality for video_transcode (0-51, default 23)
    audio_bitrate=128000 AAC target bitrate for audio_transcode (default 128000)
    audio_map=N         Select ONLY the Nth audio track (1-indexed, by PMT
                        order) — every other audio track is dropped from
                        output entirely. Default (unset): select and
                        process ALL audio tracks found in the PMT — each
                        one independently transcoded to AAC if it's
                        MP2/MP3, passed through untouched if already AAC,
                        or passed through with a warning if it's a type
                        we can't decode (e.g. AC3).
    audio_coder=twoloop AAC encoder algorithm: "twoloop" (default, best
                        quality/bit) or "fast" (~2.4x lower CPU, slightly
                        less optimal bit allocation — measured via direct
                        benchmark, not a guess). Native ffmpeg AAC encoder
                        only — libfdk_aac is not available in this build,
                        and has no VBR mode (CBR/bit_rate only).

  EXAMPLES:
    # JAYA TV HD (H264 + MP2, interlaced 1080i):
    236.1.3.78 1026 /hls/jaya jaya 192.168.82.2 - - - 0  audio_transcode=1 fix_interlace=1

    # STAR MOVIES HD (H264 + MP2):
    237.1.1.17 1001 /hls/star star 192.168.81.2 - - - 0  audio_transcode=1

    # Old MPEG-2 SD channel:
    235.1.1.10 1000 /hls/old  old  192.168.81.2 - - - 0  audio_transcode=1 video_transcode=1 video_crf=23

  NOTE — this build is the original MP2Resync baseline with ONLY the
  automatic interface resolution feature grafted in (pass "-" for
  <iface> and the box figures out which NIC to join on per channel —
  see probe_iface_for_group()/resolve_iface_for_group()/auto_iface()
  and their use in sock_open(), plus the matching "-"-to-empty
  normalization added to the direct-CLI-argument parsing path in
  main()). sps_patch_interlace() (fix_interlace=1) is intentionally
  left as the original, unmodified baseline version — no SEI
  pic_struct neutralizer, no SPS bit-walk fixes — by request.
*/
#define _GNU_SOURCE
#include <stdio.h>
#include <stdlib.h>
#include <stdint.h>
#include <string.h>
#include <unistd.h>
#include <errno.h>
#include <pthread.h>
#include <signal.h>
#include <fcntl.h>
#include <time.h>
#include <poll.h>
#include <ifaddrs.h>
#include <net/if.h>
#include <sys/stat.h>
#include <sys/file.h>
#include <sys/socket.h>
#include <arpa/inet.h>
#include <netinet/in.h>
#include <openssl/aes.h>
#include <mpg123.h>
#include <libavcodec/avcodec.h>
#ifndef FF_PROFILE_H264_MAIN
#define FF_PROFILE_H264_MAIN AV_PROFILE_H264_MAIN
#endif
#include <libavutil/channel_layout.h>
#include <libavutil/frame.h>
#include <libavutil/opt.h>
#include <libavutil/imgutils.h>
#include <libavutil/log.h>
#include <libswresample/swresample.h>
#include <libswscale/swscale.h>
#include <math.h>

/* -- tunables ---------------------------------------------------- */
#define SEGMENT_SECS   2
/* Hard ceiling on how long a single segment is allowed to stay open
 * while waiting for the next has_keyframe()/IDR cut point in simple-
 * copy/audio_transcode mode (see the segment-cut block in pkt_process).
 * Root cause this guards against, confirmed via real production log
 * analysis: this source's encoder occasionally goes far longer than
 * its normal cadence between functional-keyframe NALs (no CC drop
 * involved — cc_tot stayed flat across the whole event both times it
 * was observed), and without ANY timeout the segment just keeps
 * growing — confirmed real cases: two genuine 14.2-second segments
 * (7x the 2s target) with zero trace in cc_tot/idle/STALE, since none
 * of those track "expected keyframe is late", only "no packets at
 * all". A real player has no equivalent patience — this is exactly
 * the kind of gap that shows up as a visible freeze/black frame on
 * playback. Set well above SEGMENT_SECS so it only fires on a genuine
 * anomaly, never on normal jitter (the existing 10% tolerance already
 * handles that).                                                      */
#define MAX_SEG_WAIT_SECS (SEGMENT_SECS*4)
/* Ceiling on how long a genuine wall-clock signal gap (stale-detection
 * age>=SEGMENT_SECS*2, see the stale-detection block in nic_thread) is
 * bridged in place with PCR-interpolated stuffing before escalating to
 * the full seg_close()/PAT-re-acquisition/CLEAN START recovery path.
 * Set comfortably above the ~4.0-4.05s gaps consistently observed in
 * real production logs on this source (confirmed via multiple STALE
 * events all measuring within 5ms of each other -- a suspiciously
 * exact, likely upstream-scheduled interruption, not random loss) so
 * those bridge cleanly with margin, while a genuinely extended outage
 * (source actually down, not a brief scheduled blip) still gets the
 * full re-acquisition it actually needs rather than bridging dead air
 * indefinitely.                                                        */
#define BRIDGE_GAP_MAX_SECS 10.0
#define MAX_SEGMENTS   15
#define KEEP_EXTRA     5
#define MAX_CHANNELS   512
#define MAX_NICS       16
#define WRITE_BATCH    128
#define RCVBUF         (32<<20)
#define DUR_SLOTS      256
#define MAX_PER_CHAN   7
#define RECV_RING_PKTS 4096 /* per-channel recv/process decoupling ring
    capacity, in TS packets (188B each -> ~770KB/channel). Sized to
    absorb several seconds of backlog at typical bitrates so a slow
    pkt_process() burst (audio_transcode's mpg123/AAC path in
    particular) can never block recv() from draining the kernel socket
    buffer -- see nic_thread for the full rationale. Tune down if
    per-channel memory is tight at high channel counts, tune up if
    ring_drops ever shows nonzero in the SIGUSR1 dump on a box that
    still has CPU headroom.                                            */
#define TS_SZ          188

/* -- PCR-aware gap stuffing — matches TSDuck's null-packet approach --
 *
 * When a CC discontinuity is detected on the video/PCR PID in simple-
 * copy mode, we need to fill the gap with packets that maintain a
 * smooth, continuous PCR clock — not just null packets with PID=0x1FFF.
 *
 * Why PCR continuity matters: ExoPlayer (and all HLS-compliant players)
 * use the PCR as the master clock reference for A/V sync and buffer
 * management. If N packets are missing and we just write N PID=0x1FFF
 * null packets in their place, the PCR value in the NEXT REAL PACKET
 * (written by the encoder for wall-clock time T+N_packets) arrives
 * immediately after the nulls with NO time gap in the file — the
 * player's PCR clock jumps forward abruptly by the gap duration
 * (~1.91ms for a 7-packet gap at 5.5Mbps). This causes the decoder
 * to rush its output to catch up, producing visible drops/stutter.
 *
 * TSDuck solves this by writing the null packets with INTERPOLATED
 * PCR VALUES on the actual PCR PID (same PID as video, 0x0641 for
 * this stream) using adaptation-field-only packets (AFC=2). The PCR
 * advances smoothly across the gap so the decoder sees a continuous
 * clock and conceals the missing frames silently.
 *
 * Confirmed by direct comparison: same source through TSDuck?ffmpeg
 * plays clean in ExoPlayer, same source through udp_hls with plain
 * null stuffing still shows drops — the PCR discontinuity is the
 * remaining difference.
 *
 * ts_read_pcr(): extract the 27MHz PCR value from a TS packet that
 *   has an adaptation field with PCR_flag set (byte 5 bit 4 = 1).
 *   Returns -1 if this packet has no PCR.
 *
 * ts_write_pcr_pkt(): build a complete 188-byte adaptation-field-only
 *   TS packet on the given PID carrying the given 27MHz PCR value.
 *   Used to stuff interpolated PCR packets into the gap.             */

static int64_t ts_read_pcr(const uint8_t*p){
    /* TS header: p[0]=sync, p[1..2]=flags+PID, p[3]=CC+AFC
     * AFC field: bits 5-4 of p[3] = adaptation_field_control
     *   0x20 = AFC=2 (adaptation only)
     *   0x30 = AFC=3 (adaptation + payload)
     * adaptation_field starts at p[4]:
     *   p[4] = adaptation_field_length
     *   p[5] = flags byte: bit 4 = PCR_flag
     *   p[6..11] = PCR if PCR_flag set (48 bits total):
     *     base[32:0] in bits 47-15, marker=1 in bit 14 (actually
     *     the standard encoding is: base[32:25] in p[6],
     *     base[24:17] in p[7], base[16:9] in p[8],
     *     base[8:1] in p[9], base[0] in bit7 of p[10],
     *     reserved 6 bits (0x7E), ext[8] in bit0 of p[10],
     *     ext[7:0] in p[11].
     *     27MHz PCR = base*300 + ext                               */
    int afc=(p[3]>>4)&0x3;
    if(afc!=2&&afc!=3)return -1;        /* no adaptation field */
    if(p[4]<7)return -1;                /* adaptation field too short */
    if(!(p[5]&0x10))return -1;         /* PCR_flag not set */
    uint64_t base=(uint64_t)p[6]<<25|(uint64_t)p[7]<<17|
                  (uint64_t)p[8]<<9|(uint64_t)p[9]<<1|
                  ((p[10]>>7)&1);
    uint16_t ext=(uint16_t)((p[10]&1)<<8)|p[11];
    return (int64_t)(base*300+ext);}

static void ts_write_pcr_pkt(uint8_t*out,uint16_t pid,int64_t pcr27){
    /* Build a 188-byte adaptation-field-only TS packet (AFC=2) on
     * the given PID, carrying the given 27MHz PCR value.
     * Layout: 4-byte header + 184-byte adaptation field.
     * adaptation_field_length = 183 (rest of packet after the length
     * byte itself), flags byte with PCR_flag=1, 6 PCR bytes, then
     * stuffing bytes (0xFF) to fill the rest.                       */
    memset(out,0xFF,188);
    out[0]=0x47;
    out[1]=(uint8_t)(0x00|((pid>>8)&0x1F));
    out[2]=(uint8_t)(pid&0xFF);
    out[3]=0x20;                         /* AFC=2, CC=0 (no payload) */
    out[4]=183;                          /* adaptation_field_length  */
    out[5]=0x10;                         /* PCR_flag=1, rest=0       */
    uint64_t base=(uint64_t)pcr27/300;
    uint16_t ext=(uint16_t)(pcr27%300);
    out[6]=(uint8_t)((base>>25)&0xFF);
    out[7]=(uint8_t)((base>>17)&0xFF);
    out[8]=(uint8_t)((base>>9)&0xFF);
    out[9]=(uint8_t)((base>>1)&0xFF);
    out[10]=(uint8_t)(((base&1)<<7)|0x7E|((ext>>8)&1));
    out[11]=(uint8_t)(ext&0xFF);
    /* bytes 12..187 are already 0xFF (stuffing) from memset        */}

/* Null TS packet — PID=0x1FFF, payload-only, all-0xFF payload.
 * Used for non-PCR-PID gap filling (audio PID drops etc.) where
 * PCR restamping is not needed.                                      */
static const uint8_t NULL_TS_PKT[188]={
    0x47,0x1F,0xFF,0x10, /* sync, PID=0x1FFF, payload_only, CC=0 */
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF, /* 184 bytes of payload */
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,
    0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF,0xFF};
#define MAX_AUD_TRACKS 8       /* max simultaneously selected/transcoded
                                   audio tracks per channel - real
                                   broadcasts essentially never exceed
                                   a handful of audio tracks            */
#define AES_BLK        16
#define UDP_MAX        65536

/* -- utilities --------------------------------------------------- */
static void ts_now(char *b){
    time_t t=time(NULL); struct tm *m=localtime(&t);
    snprintf(b,9,"%02d:%02d:%02d",m->tm_hour,m->tm_min,m->tm_sec);
}
static inline double wall_el(const struct timespec *a,const struct timespec *b){
    return (b->tv_sec-a->tv_sec)+(b->tv_nsec-a->tv_nsec)/1e9;
}

/* -- lock-free event ring ---------------------------------------- */
#define EVT_RING 256
typedef enum{ EVT_NONE=0,EVT_PAT,EVT_PMT,EVT_START,
              EVT_OPEN,EVT_CLOSE,EVT_CC,EVT_STALE }EvtType;
typedef struct{ EvtType type;time_t when;uint64_t seq;
                uint32_t u32a,u32b;uint16_t u16a;uint8_t u8a,u8b;}Evt;
typedef struct{ Evt ring[EVT_RING];unsigned wr,rd;}EvtRing;
static inline void evpush(EvtRing*r,const Evt*e){
    unsigned w=__atomic_load_n(&r->wr,__ATOMIC_RELAXED);
    r->ring[w&(EVT_RING-1)]=*e;
    __atomic_store_n(&r->wr,w+1,__ATOMIC_RELEASE);}
static void evdrain(EvtRing*r,const char*n){
    unsigned w=__atomic_load_n(&r->wr,__ATOMIC_ACQUIRE);
    while(r->rd!=w){
        Evt*e=&r->ring[r->rd&(EVT_RING-1)];r->rd++;
        struct tm*m=localtime(&e->when);char t[9];
        snprintf(t,9,"%02d:%02d:%02d",m->tm_hour,m->tm_min,m->tm_sec);
        switch(e->type){
        case EVT_PAT:  printf("  [%s][%s] PAT pmt_pid=0x%04X\n",t,n,e->u16a);break;
        case EVT_PMT:  printf("  [%s][%s] PMT vid=0x%04X aud=0x%04X pcr=0x%04X\n",t,n,e->u16a,e->u32a,e->u32b);break;
        case EVT_START:printf("  [%s][%s] CLEAN START vid=0x%04X\n",t,n,e->u16a);break;
        case EVT_OPEN: printf("  [%s][%s] opened  index%llu.ts\n",t,n,(unsigned long long)e->seq);break;
        case EVT_CLOSE:printf("  [%s][%s] closed  index%llu.ts  %u.%03us  %ukbps\n",
                              t,n,(unsigned long long)e->seq,e->u32a/1000,e->u32a%1000,e->u32b);break;
        case EVT_CC:   printf("  [%s][%s] CC DROP pid=0x%04X exp=0x%X got=0x%X\n",t,n,e->u16a,e->u8a,e->u8b);break;
        case EVT_STALE:printf("  [%s][%s] STALE no input for %u.%03us\n",t,n,e->u32a/1000,e->u32a%1000);break;
        default:break;}}}

/* ----------------------------------------------------------------
   AUDIO TRANSCODER  MP2/MP3 -> AAC-LC
   mpg123 decode -> PCM -> libavcodec AAC encode -> ADTS -> TS
   ---------------------------------------------------------------- */
#define AAC_FRAME  1024
#define ADTS_HDR   7
static const int SR_TABLE[]={96000,88200,64000,48000,44100,32000,
                              24000,22050,16000,12000,11025,8000,7350};
typedef enum{AUDXC_SRC_MP2=0,AUDXC_SRC_AC3}AudXcSrc;
typedef struct{
    int             active;
    AudXcSrc        src_codec;    /* which decoder front-end is in use   */
    mpg123_handle  *mh;           /* MP2/MP3 front-end (src_codec==MP2)  */
    AVCodecContext *ac3_dec_ctx;  /* AC3/E-AC3 front-end (src_codec==AC3)*/
    AVPacket       *ac3_pkt;
    AVFrame        *ac3_frame;
    AVCodecParserContext *ac3_parser; /* finds AC3 frame boundaries across
        TS packets — a single AC3 frame typically spans several TS
        packets (confirmed: ~8 packets at 384kbps/48kHz), so feeding one
        packet's payload as "one complete frame" (the initial, broken
        attempt) fails decode with "Invalid data found" almost every
        time. The parser accumulates bytes across calls and hands back
        exactly one complete frame at a time, regardless of input
        chunking — the correct, standard way to handle this.          */
    struct SwrContext *swr;       /* fltp[+downmix] -> interleaved s16,
        allocated lazily once the AC3 stream's actual channel layout/
        sample rate is known from the first decoded frame             */
    int             swr_ready;
    AVCodecContext *ctx;
    AVFrame        *frame;
    AVPacket       *pkt;
    int             frame_size,sr_idx;
    int16_t         pcm[AAC_FRAME*2*4];
    int             pcm_fill;
    uint8_t         out[TS_SZ*32];/* 32 TS pkts: handles large AAC frames */
    /* adts_scratch/pcm_scratch/mp2_raw_scratch used to live HERE, one
     * copy per AudXc (i.e. per audio track, up to MAX_AUD_TRACKS=8 per
     * channel). Moved up to Ch (see Ch.aud_scratch) and shared across
     * all of a channel's tracks instead: audxc_push for a given
     * channel is always called sequentially from pkt_process's single-
     * threaded per-packet dispatch -- never two tracks' audxc_push
     * calls in flight at once for the SAME channel -- so one shared
     * set of scratch buffers per channel is exactly as safe as one per
     * track, at 1/8th the static memory cost (different channels run
     * on different NIC threads and each get their own Ch, hence their
     * own scratch buffers -- this is per-channel sharing, not a true
     * global). See AudScratch doc at its definition for full detail.  */
    int             out_len;
    int64_t         pts,pts_inc;
    int             pts_ok;
    int             pusi_seen;   /* discard continuation before first PUSI */
    uint64_t        push_calls;     /* total audxc_push calls with real
        data (srclen>0) since this slot was initialized - for the
        SIGUSR1 dump, to distinguish "genuinely stalled despite
        active=1" from other failure modes. Confirmed necessary: a
        real production dump showed active=1/pts_ok=1/pusi_seen=1
        (all healthy-looking) at the exact same moment 7 consecutive
        real segments showed zero audio packets.                      */
    uint64_t        push_calls_with_output; /* subset of push_calls
        where audxc_push actually returned nonzero output bytes.      */
    int             stall_calls; /* consecutive audxc_push calls with
        zero output — see the logging in audxc_push for how this is
        used (mirrors VidXc.stall_calls' existing pattern for video).
        Added specifically to get direct proof of whether audio is
        silently stuck after a STALE/CLEAN START recovery, instead of
        inferring it from an empty track downstream.                  */
    int64_t         last_seen_source_pts; /* most recent real PES PTS
        seen on this track (updated on every PUSI, not just the
        first) - see latency_comp doc for how this is used.          */
    int             last_seen_source_pts_valid;
    int64_t         samples_since_anchor; /* total PCM samples fed into
        the AAC encoder since pts_ok was set - used to compute the
        EXPECTED output pts at any point, for comparison against
        last_seen_source_pts (see latency_comp doc).                  */
    int             latency_measured;  /* 1 once latency_comp has been
        computed from a real measurement - measured once per channel
        start/CC-recovery, not every frame, since pipeline latency is
        structurally fixed for a given codec/decoder configuration.   */
    int64_t         latency_comp; /* measured pipeline latency, in
        90kHz PTS units, in the SOURCE's own timestamp domain — NOT
        wall-clock time. Computed as: (anchor_pts + samples_since_
        anchor_in_90khz) - last_seen_source_pts, sampled once a
        reasonable amount of source data has passed through. This
        directly answers "how far behind the source's own clock has
        our output PTS counter drifted, due to real decode+encode
        latency" — confirmed necessary via direct measurement: video
        runs ~60-100ms ahead of every transcoded audio track (AC3 and
        MP2 alike), present ONLY on audio_transcode channels (absent
        on passthrough-AAC channels), meaning this is real, physical
        pipeline latency that must be measured in the source's own
        timestamp domain, not wall-clock time (a wall-clock-based
        measurement was tried first and confirmed, via direct
        comparison against real captured output, to produce
        essentially no correction at all — CPU processing time and
        audio-timeline latency are different quantities).             */
}AudXc;

/* Scratch buffers shared by all of ONE channel's audio tracks (see the
 * comment in AudXc above for why this is safe: a channel's audxc_push
 * calls -- across however many of its tracks are active -- are always
 * sequential, never concurrent, since they're all driven from the
 * same single-threaded per-packet pkt_process dispatch). One of these
 * lives in Ch, not in AudXc/AudSlot, so the cost is paid once per
 * channel instead of once per track (up to MAX_AUD_TRACKS=8x more
 * before this change). Different channels run on different NIC
 * threads and each have their own Ch and thus their own AudScratch --
 * this is per-channel sharing, not a true cross-channel global.       */
typedef struct{
    uint8_t  adts[ADTS_HDR+1024]; /* see audxc_feed_pcm's use: sized for
        AAC-LC up to ~320kbps with headroom (1024 samples/frame @
        48kHz: even 320kbps CBR averages ~853 bytes/frame -- see
        audio_bitrate option doc for the supported range).            */
    int16_t  pcm[AAC_FRAME*5]; /* see audxc_push_ac3's use: sized for
        the actual worst-case AC3 resample output -- max AC3 frame is
        1536 samples/channel, worst-case upsampling from a 32kHz
        source to 48kHz gives 1536*1.5=2304 samples/channel, *2 for
        stereo interleaved = 4608 int16 values; this (5120) covers
        that with ~10% headroom for swr's internal delay buffer.      */
    uint8_t  mp2_raw[1152*2*2*2]; /* see audxc_push_mp2's use: sized for
        the real spec maximum -- MPEG-1 Layer II/III max samples/frame
        = 1152, stereo int16 = 2 bytes/sample -> 4608 bytes minimum;
        this (9216) gives 2x headroom.                                */
}AudScratch;

static void adts_hdr(uint8_t*h,int pay,int sri,int ch){
    int tot=pay+ADTS_HDR;
    h[0]=0xFF;h[1]=0xF1;
    h[2]=(uint8_t)(0x40|(sri<<2)|((ch>>2)&1));
    h[3]=(uint8_t)(((ch&3)<<6)|((tot>>11)&3));
    h[4]=(uint8_t)((tot>>3)&0xFF);
    h[5]=(uint8_t)(((tot&7)<<5)|0x1F);
    h[6]=0xFC;
}

/* Write ADTS frame into one or more TS packets.
 * First packet: PUSI=1 + PES header. Continuations: PUSI=0.
 * Returns total bytes written (multiple of TS_SZ).            */
static int adts_to_ts(uint8_t*dst,uint16_t pid,
                      const uint8_t*adts,int alen,int64_t pts,uint8_t*cc){
    int written=0,src_off=0,first=1;
    int pes_hdr=14; /* 9 fixed + 5 PTS */
    while(src_off<alen){
        uint8_t*pkt=dst+written;
        memset(pkt,0,TS_SZ); /* zero, not 0xFF — see padding note below */
        pkt[0]=0x47;
        pkt[1]=(uint8_t)((first?0x40:0x00)|((pid>>8)&0x1F));
        pkt[2]=(uint8_t)(pid&0xFF);
        int po;
        if(first){
            pkt[3]=(uint8_t)(0x10|(*cc&0x0F));
            (*cc)=(*cc+1)&0x0F;
            uint8_t*p=pkt+4;
            int plen=pes_hdr-6+alen;
            p[0]=0;p[1]=0;p[2]=1;p[3]=0xC0;
            p[4]=(uint8_t)(plen>>8);p[5]=(uint8_t)(plen&0xFF);
            p[6]=0x80;p[7]=0x80;p[8]=0x05;
            p[9] =(uint8_t)(0x21|((pts>>29)&0x0E));
            p[10]=(uint8_t)((pts>>22)&0xFF);
            p[11]=(uint8_t)(0x01|((pts>>14)&0xFE));
            p[12]=(uint8_t)((pts>>7)&0xFF);
            p[13]=(uint8_t)(0x01|((pts<<1)&0xFE));
            po=4+pes_hdr; first=0;
        } else {
            po=4;
            int remain=alen-src_off;
            if(remain<TS_SZ-4){
                /* Last packet of this ADTS frame and it won't fill the TS
                 * packet exactly. The OLD code memset() the whole packet
                 * to 0xFF and just copied the short payload on top,
                 * leaving a run of raw 0xFF bytes immediately after the
                 * real ADTS data — INSIDE the payload area of an AFC=01
                 * (payload-only) packet, where the TS spec defines every
                 * byte as stream data with no legal padding mechanism.
                 * A run of 0xFF 0xFF... decodes as a fake ADTS sync word
                 * (0xFF) + flags byte (0xFF) + length-field bytes that
                 * compute to 8191 (0x1FFF, the field's max value) — and
                 * ffmpeg's demuxer, resyncing after the genuine frame
                 * ends, locks onto this fake header and tries to read an
                 * 8191-byte "frame" that doesn't exist, corrupting every
                 * subsequent AAC frame in the segment. This exact pattern
                 * was confirmed byte-for-byte in real captured output.
                 * Fix: use a proper adaptation field with spec-legal
                 * stuffing_byte padding (AFC=11), identical to the fix
                 * already applied to the video path in vidxc_write().    */
                int pad=(TS_SZ-4)-remain;
                int afl=pad-1; if(afl<0)afl=0;
                pkt[3]=(uint8_t)(0x30|(*cc&0x0F)); /* AFC=11 */
                (*cc)=(*cc+1)&0x0F;
                pkt[4]=(uint8_t)afl;
                if(afl>0){
                    pkt[5]=0x00;
                    for(int i=1;i<afl;i++)pkt[5+i]=0xFF; /* legal stuffing_byte */
                }
                po=4+1+afl;
            } else {
                pkt[3]=(uint8_t)(0x10|(*cc&0x0F)); /* AFC=01: full payload */
                (*cc)=(*cc+1)&0x0F;
            }
        }
        int sp=TS_SZ-po,cp=alen-src_off;if(cp>sp)cp=sp;
        memcpy(pkt+po,adts+src_off,cp);
        src_off+=cp; written+=TS_SZ;
    }
    return written;
}

/* Buffer interleaved stereo int16 PCM samples (ns sample-pairs starting
 * at s) into AAC_FRAME-sized chunks, encode each full chunk via the
 * shared AAC encoder context, and packetize the result into ADTS+TS
 * appended to a->out[]. Shared by both decoder front-ends (MP2 via
 * mpg123, AC3/E-AC3 via libavcodec) since the encode side is identical
 * regardless of source codec — only how PCM samples are produced
 * differs.                                                            */
static void audxc_feed_pcm(AudXc*a,AudScratch*scr,const int16_t*s,int ns,
                           uint16_t pid,uint8_t*cc){
    int so=0;
    while(so<ns){
        int sp=AAC_FRAME-a->pcm_fill,cp=ns-so;if(cp>sp)cp=sp;
        for(int i=0;i<cp;i++){
            a->pcm[(a->pcm_fill+i)*2+0]=s[(so+i)*2+0];
            a->pcm[(a->pcm_fill+i)*2+1]=s[(so+i)*2+1];
        }
        a->pcm_fill+=cp;so+=cp;
        if(a->pcm_fill>=AAC_FRAME){
            av_frame_make_writable(a->frame);
            float*L=(float*)a->frame->data[0];
            float*R=(float*)a->frame->data[1];
            for(int i=0;i<AAC_FRAME;i++){
                L[i]=a->pcm[i*2+0]/32768.0f;
                R[i]=a->pcm[i*2+1]/32768.0f;
            }
            a->pcm_fill=0;a->samples_since_anchor+=AAC_FRAME;
            if(!a->pts_ok){
                a->pts=0;a->pts_ok=1;
                /* Diagnostic: this fallback firing at all means the
                 * first AAC frame was ready to flush BEFORE a real
                 * source PES PTS had been captured (see the other
                 * anchor at "audio PTS anchor" in pkt_process) — a
                 * race between audio decode/flush timing and PES
                 * arrival. If this fires, audio's timeline starts at
                 * literal 0 instead of the source's real PTS, which
                 * is its own distinct route to the same class of
                 * cross-track offset bug this instrumentation is
                 * chasing.                                            */
                fprintf(stderr,"[audxc] WARNING: PTS anchor fell back to "
                        "0 — first AAC frame flushed before a real "
                        "source PTS was captured\n");
            }
            if(!a->latency_measured&&a->last_seen_source_pts_valid){
                /* Measure real pipeline latency entirely in the
                 * source's own PTS domain — never wall-clock time (a
                 * wall-clock measurement was tried first and confirmed,
                 * via direct comparison against real captured output,
                 * to produce essentially no correction at all on a
                 * normally-loaded system, since CPU processing time
                 * and audio-timeline latency are different
                 * quantities). a->pts is the real absolute source PTS
                 * captured at anchor time; samples_since_anchor is how
                 * much output-time we've produced since then. If the
                 * pipeline had zero latency, the source's clock would
                 * now read exactly a->pts + that many 90kHz units. The
                 * ACTUAL source clock (last_seen_source_pts, updated on
                 * every PUSI since) has advanced further than that —
                 * the difference is the real, physical decode+encode
                 * latency this output frame is behind by. Confirmed
                 * necessary via direct measurement: video runs a
                 * consistent ~60-100ms ahead of every transcoded audio
                 * track (AC3 and MP2 alike), present ONLY on
                 * audio_transcode channels.                            */
                int64_t expected_src_pts=a->pts+
                    (a->samples_since_anchor*90000LL/48000);
                int64_t gap=a->last_seen_source_pts-expected_src_pts;
                if(gap>0&&gap<90000){ /* sanity bound: real pipeline
                    latency should never be anywhere near a full
                    second — guards against a bogus/wrapped source PTS
                    producing a nonsensical correction               */
                    a->latency_comp=gap;
                    a->pts+=a->latency_comp; /* shift the base FORWARD:
                        this output is genuinely `gap` 90kHz-units
                        behind where the source's real clock already
                        is, so its correct presentation time is later
                        than the raw anchor would suggest — matching
                        how much decode+encode latency actually
                        elapsed before this frame became available.   */
                }
                a->latency_measured=1;
            }
            /* NOTE: ongoing (not just one-time) audio clock drift
             * correction is applied elsewhere, in pkt_process, on
             * every audio PUSI — see the comment at
             * "s->audxc.last_seen_source_pts=src_pts" below in the
             * audio_transcode PUSI handler. Doing it there (driven by
             * real PES PTS arrivals) rather than here (driven by AAC
             * frame cadence) means the correction tracks the actual
             * source clock updates directly, with no extra bookkeeping
             * needed in this function.                                 */
            a->frame->pts=a->pts; a->pts+=a->pts_inc;
            if(avcodec_send_frame(a->ctx,a->frame)<0)continue;
            while(avcodec_receive_packet(a->ctx,a->pkt)==0){
                uint8_t*adts=scr->adts;
                int pay=a->pkt->size;
                if(pay>(int)sizeof(scr->adts)-ADTS_HDR)pay=(int)sizeof(scr->adts)-ADTS_HDR;
                adts_hdr(adts,pay,a->sr_idx,2);
                memcpy(adts+ADTS_HDR,a->pkt->data,pay);
                int alen=ADTS_HDR+pay;
                if(a->out_len+TS_SZ*3<=(int)sizeof(a->out)){
                    int wrote=adts_to_ts(a->out+a->out_len,pid,
                                         adts,alen,a->frame->pts,cc);
                    a->out_len+=wrote;
                }
                av_packet_unref(a->pkt);
            }
        }
    }
}

static int audxc_init(AudXc*a,int bitrate,const char*coder,AudXcSrc src_codec){
    memset(a,0,sizeof*a);
    a->src_codec=src_codec;
    if(src_codec==AUDXC_SRC_MP2){
        mpg123_init();
        int err; a->mh=mpg123_new(NULL,&err);
        if(!a->mh)return 0;
        mpg123_param(a->mh,MPG123_FLAGS,MPG123_QUIET,0);
        mpg123_format_none(a->mh);
        mpg123_format(a->mh,48000,MPG123_STEREO,MPG123_ENC_SIGNED_16);
        if(mpg123_open_feed(a->mh)!=MPG123_OK){mpg123_delete(a->mh);return 0;}
    } else { /* AUDXC_SRC_AC3 */
        const AVCodec*dec=avcodec_find_decoder(AV_CODEC_ID_AC3);
        if(!dec)dec=avcodec_find_decoder(AV_CODEC_ID_EAC3);
        if(!dec)return 0;
        a->ac3_dec_ctx=avcodec_alloc_context3(dec);
        if(avcodec_open2(a->ac3_dec_ctx,dec,NULL)<0){
            avcodec_free_context(&a->ac3_dec_ctx);return 0;}
        a->ac3_pkt=av_packet_alloc();
        a->ac3_frame=av_frame_alloc();
        a->ac3_parser=av_parser_init((int)dec->id);
        if(!a->ac3_parser){
            avcodec_free_context(&a->ac3_dec_ctx);
            av_packet_free(&a->ac3_pkt);av_frame_free(&a->ac3_frame);
            return 0;}
    }
    const AVCodec*codec=avcodec_find_encoder(AV_CODEC_ID_AAC);
    if(!codec){
        if(a->mh)mpg123_delete(a->mh);
        if(a->ac3_dec_ctx)avcodec_free_context(&a->ac3_dec_ctx);
        if(a->ac3_pkt)av_packet_free(&a->ac3_pkt);
        if(a->ac3_frame)av_frame_free(&a->ac3_frame);
        return 0;}
    a->ctx=avcodec_alloc_context3(codec);
    a->ctx->sample_rate=48000; a->ctx->bit_rate=bitrate>0?bitrate:128000;
    av_channel_layout_default(&a->ctx->ch_layout,2); /* AAC output is
        always stereo regardless of source — AC3 5.1 sources are
        downmixed to stereo when PCM is extracted in audxc_push.      */
    a->ctx->sample_fmt=AV_SAMPLE_FMT_FLTP; /* all AAC encoders use FLTP */
    if(coder&&coder[0])av_opt_set(a->ctx->priv_data,"aac_coder",coder,0);
    if(avcodec_open2(a->ctx,codec,NULL)<0){
        avcodec_free_context(&a->ctx);
        if(a->mh)mpg123_delete(a->mh);
        if(a->ac3_dec_ctx)avcodec_free_context(&a->ac3_dec_ctx);
        if(a->ac3_pkt)av_packet_free(&a->ac3_pkt);
        if(a->ac3_frame)av_frame_free(&a->ac3_frame);
        return 0;}
    a->frame_size=a->ctx->frame_size;
    a->frame=av_frame_alloc();
    a->frame->nb_samples=a->frame_size;
    a->frame->format=a->ctx->sample_fmt;
    av_channel_layout_copy(&a->frame->ch_layout,&a->ctx->ch_layout);
    av_frame_get_buffer(a->frame,0);
    a->pkt=av_packet_alloc();
    a->sr_idx=3; /* 48kHz */
    for(int i=0;i<(int)(sizeof SR_TABLE/sizeof*SR_TABLE);i++)
        if(SR_TABLE[i]==48000){a->sr_idx=i;break;}
    a->pts_inc=(int64_t)a->frame_size*90000/48000; /* =1920 */
    a->active=1;
    if(src_codec==AUDXC_SRC_AC3)
        printf("[audxc] AC3->AAC 48kHz stereo (downmixed if needed) %dkbps ready\n",
               bitrate>0?bitrate/1000:128);
    else
        printf("[audxc] MP2->AAC 48kHz stereo 128kbps ready\n");
    return 1;
}
static void audxc_free(AudXc*a){
    if(!a->active)return;
    av_frame_free(&a->frame);av_packet_free(&a->pkt);
    avcodec_free_context(&a->ctx);
    if(a->src_codec==AUDXC_SRC_MP2){
        mpg123_delete(a->mh);mpg123_exit();
    } else {
        av_frame_free(&a->ac3_frame);av_packet_free(&a->ac3_pkt);
        avcodec_free_context(&a->ac3_dec_ctx);
        if(a->ac3_parser)av_parser_close(a->ac3_parser);
        if(a->swr)swr_free(&a->swr);
    }
    a->active=0;
}
/* Feed raw MP2 payload bytes, produce AAC TS packets in a->out[].
 * Returns bytes written (multiple of TS_SZ).                  */
/* MP2/MP3 source: existing mpg123-based decode, now feeding the shared
 * audxc_feed_pcm helper instead of inline duplicate buffering/encode
 * code (no behavior change from before — pure refactor).              */
static int audxc_push_mp2(AudXc*a,AudScratch*scr,const uint8_t*mp2,int mp2len,
                          uint16_t pid,uint8_t*cc){
    mpg123_feed(a->mh,mp2,(size_t)mp2len);
    uint8_t*raw=scr->mp2_raw;
    size_t done; int ret;
    while((ret=mpg123_read(a->mh,raw,sizeof scr->mp2_raw,&done))==MPG123_OK
          ||ret==MPG123_NEW_FORMAT||ret==MPG123_NEED_MORE){
        if(!done){
            /* MPG123_NEW_FORMAT fires with done=0 the instant mpg123 locks
             * sync — the PCM for the frame that triggered the lock is
             * fetched on the *next* read call, not this one. Don't break;
             * loop again immediately to drain it in the same push.        */
            if(ret==MPG123_NEW_FORMAT) continue;
            /* MPG123_NEED_MORE with done=0: truly no more PCM buffered,
             * decoder is waiting for the next mpg123_feed(). Stop here. */
            break;
        }
        /* CRITICAL: MPG123_NEED_MORE can carry done>0 — mpg123 hands back
         * PCM for a fully-decoded frame in the SAME call that also says
         * "I need more input for the *next* frame". Discarding this case
         * (old code's loop condition excluded NEED_MORE entirely) silently
         * dropped every other frame's audio — net result was zero AAC ever
         * produced. We must process this data exactly like MPG123_OK,
         * then stop (no more buffered data left after a NEED_MORE).      */
        int16_t*s=(int16_t*)raw;
        int ns=(int)(done/sizeof(int16_t))/2;
        audxc_feed_pcm(a,scr,s,ns,pid,cc);
    }
    /* MPG123_ERR (or any return code outside the loop's accepted set)
     * means the decoder's internal stream state is desynced — typically
     * triggered by real upstream packet loss (a CC drop) corrupting the
     * MPEG audio frame boundary mpg123 was tracking. Confirmed as the
     * root cause of a real production incident via direct SIGUSR1
     * dump comparison: push_calls climbed by 14252 over 96s/48 segments
     * while push_calls_with_output, write_loop_entries and audxc.pts
     * stayed perfectly flat the entire time — mpg123_feed() never
     * errors (it only appends to its internal buffer), so packets kept
     * arriving and being fed in, but mpg123_read() was silently
     * returning MPG123_ERR on every single call forever after, with no
     * path back to a working decode state. The PMT kept advertising an
     * audio track throughout, and a production segment from this same
     * incident showed the audio PID completely absent (0 packets out of
     * ~6900 TS packets, verified by direct binary parse), matching the
     * ~96s of total silence. Closing and reopening the feed
     * (mpg123_close/mpg123_open_feed) resets mpg123's internal sync
     * state without tearing down the whole AudXc (no need to reallocate
     * the AAC encoder context, which is independent of the MP2 decode
     * front-end) — exactly the same "force a resync after real packet
     * loss" approach already used on the video CC-drop path
     * (force_idr+au_reset), just applied to the MP2 decoder instead of
     * the H264 decoder. We deliberately do NOT retry mpg123_read()
     * again in this same call: the loop above has already drained
     * everything decodable from this feed; any bytes still sitting in
     * the now-closed handle's internal buffer are discarded along with
     * it, which is correct — they belong to the corrupted stream state
     * we are resyncing away from, identical to how a video CC drop
     * discards the in-flight AU rather than trying to salvage it.      */
    if(ret==MPG123_ERR){
        static time_t _last_warn=0; time_t now_t=time(NULL);
        if(now_t!=_last_warn){
            fprintf(stderr,"[audxc] MPG123_ERR — MP2 decoder desynced "
                    "(likely real upstream packet loss), resetting feed\n");
            _last_warn=now_t;}
        mpg123_close(a->mh);
        if(mpg123_open_feed(a->mh)!=MPG123_OK){
            /* Reopen itself failing is a deeper problem (e.g. mpg123
             * internal allocation failure) — mark this track inactive
             * rather than spin forever calling a handle that can't even
             * accept feed data.                                         */
            fprintf(stderr,"[audxc] mpg123_open_feed FAILED on reset — "
                    "disabling this audio track\n");
            a->active=0;
        }
        /* pts_ok/pusi_seen/latency_measured/samples_since_anchor reset
         * exactly like a CC-drop recovery (see pkt_process's audio CC
         * handling) — the anchor PTS this track had is now meaningless
         * since the encoder will resume from a fresh PCM stream with no
         * continuity to what came before.                                */
        a->pts_ok=0; a->pusi_seen=0;
        a->last_seen_source_pts_valid=0;
        a->latency_measured=0; a->samples_since_anchor=0;
        a->pcm_fill=0; /* CRITICAL — was missing here: pcm[] is a
            partial-frame accumulator waiting to reach AAC_FRAME (1024)
            samples before encoding. Without resetting pcm_fill, fresh
            post-resync PCM gets appended onto whatever stale/partial
            data was sitting in the buffer at the moment of the desync,
            corrupting the next several AAC frames right at the point
            of recovery — real, confirmed cause of audio staying
            silently broken even after mpg123 itself resyncs cleanly.  */
    }
    return a->out_len;
}

/* AC3/E-AC3 source: decode via libavcodec, downmix to stereo (if the
 * source is 5.1/surround) and convert fltp->s16 via swresample, then
 * feed the shared PCM pipeline. Unlike MP2/mpg123 (a streaming decoder
 * fed arbitrary-sized chunks), AC3 is frame-oriented — each TS PES
 * payload we're handed here is expected to be one complete AC3 sync
 * frame (this matches how the caller in pkt_process delivers it: a
 * full audio PES payload per call, same as the MP2 path receives).    */
static int audxc_push_ac3(AudXc*a,AudScratch*scr,const uint8_t*ac3,int ac3len,
                          uint16_t pid,uint8_t*cc){
    const uint8_t*in_data=ac3;int in_len=ac3len;
    while(in_len>0){
        uint8_t*frame_data=NULL;int frame_size=0;
        int consumed=av_parser_parse2(a->ac3_parser,a->ac3_dec_ctx,
            &frame_data,&frame_size,in_data,in_len,
            AV_NOPTS_VALUE,AV_NOPTS_VALUE,0);
        if(consumed<0)break; /* genuine parser error — stop, avoid
                                 an infinite loop on malformed input.
                                 consumed==0 is NOT an error: it means
                                 the parser handed back a complete frame
                                 it had already buffered, without
                                 needing to consume any new bytes this
                                 call — that frame must still be
                                 processed below, not discarded. (This
                                 distinction was the actual bug found
                                 via direct trace: real, valid AC3
                                 frames — confirmed 1536 bytes, matching
                                 384kbps/48kHz exactly — were being
                                 silently discarded every time because
                                 consumed==0 was wrongly treated as a
                                 stuck/error state and broke the loop
                                 before the frame was ever decoded.)    */
        in_data+=consumed;in_len-=consumed;
        if(frame_size<=0){
            if(consumed==0)break; /* truly no progress AND no frame —
                                      now it's safe to stop            */
            continue; /* parser consumed bytes but hasn't accumulated
                          a full frame yet — normal, keep feeding      */
        }
        av_packet_unref(a->ac3_pkt);
        /* Point directly at the parser's own buffer instead of
         * allocating a fresh one and copying into it — frame_data is
         * valid for the duration of this call, and avcodec_send_packet
         * only reads from the packet synchronously without retaining
         * it afterward, so no copy is actually needed here. This
         * removes one malloc+memcpy+free cycle per AC3 frame (~31x/sec
         * per active AC3/E-AC3 track at common bitrates) — a real,
         * measurable cost when multiple audio tracks run alongside
         * video decode/encode on this codebase's single-threaded
         * per-channel design.                                         */
        a->ac3_pkt->data=frame_data;a->ac3_pkt->size=frame_size;
        if(avcodec_send_packet(a->ac3_dec_ctx,a->ac3_pkt)<0)continue;
        while(avcodec_receive_frame(a->ac3_dec_ctx,a->ac3_frame)==0){
            if(!a->swr_ready){
                /* Lazily build the resampler now that we know the AC3
                 * stream's actual channel layout and sample rate — AC3
                 * streams can be mono, stereo, 5.1, etc, and this isn't
                 * known until the first frame is decoded.              */
                AVChannelLayout out_layout;
                av_channel_layout_default(&out_layout,2); /* stereo out */
                int rc=swr_alloc_set_opts2(&a->swr,
                    &out_layout,AV_SAMPLE_FMT_S16,48000,
                    &a->ac3_frame->ch_layout,(enum AVSampleFormat)a->ac3_frame->format,
                    a->ac3_frame->sample_rate,0,NULL);
                if(rc>=0&&a->swr)rc=swr_init(a->swr);
                av_channel_layout_uninit(&out_layout);
                if(rc<0||!a->swr){av_frame_unref(a->ac3_frame);continue;}
                a->swr_ready=1;
                printf("[audxc] AC3 source: %d ch, %dHz -> downmixing to stereo 48kHz\n",
                       a->ac3_frame->ch_layout.nb_channels,a->ac3_frame->sample_rate);
            }
            /* swr_convert may need to resample (if source isn't 48kHz) —
             * size the output buffer generously for that case.         */
            int max_out_samples=av_rescale_rnd(
                swr_get_delay(a->swr,a->ac3_frame->sample_rate)+a->ac3_frame->nb_samples,
                48000,a->ac3_frame->sample_rate,AV_ROUND_UP);
            int16_t*pcmbuf=scr->pcm; /* persistent, shared per-channel scratch */
            if(max_out_samples*2>(int)(sizeof(scr->pcm)/sizeof(int16_t)))
                max_out_samples=(int)(sizeof(scr->pcm)/sizeof(int16_t))/2;
            uint8_t*out_planes[1]={(uint8_t*)pcmbuf};
            int got=swr_convert(a->swr,out_planes,max_out_samples,
                                (const uint8_t**)a->ac3_frame->extended_data,
                                a->ac3_frame->nb_samples);
            av_frame_unref(a->ac3_frame);
            if(got>0)audxc_feed_pcm(a,scr,pcmbuf,got,pid,cc);
        }
    }
    return a->out_len;
}

static int audxc_push(AudXc*a,AudScratch*scr,const uint8_t*src,int srclen,
                      uint16_t pid,uint8_t*cc){
    a->out_len=0;
    if(!a->active||srclen<=0)return 0;
    a->push_calls++;
    int result=(a->src_codec==AUDXC_SRC_MP2)?
        audxc_push_mp2(a,scr,src,srclen,pid,cc):
        audxc_push_ac3(a,scr,src,srclen,pid,cc);
    if(result>0){
        a->push_calls_with_output++;
        if(a->stall_calls>=50){
            /* Was stalled (50+ consecutive pushes with zero output —
             * at typical MP2 packet cadence that's several real
             * seconds of silence) and just recovered on its own.
             * Logging this explicitly, with the exact counts, turns
             * "audio came back eventually" into hard evidence instead
             * of something inferred from the stats line looking
             * normal again a few segments later.                     */
            printf("[audxc] recovered after %d consecutive stalled "
                   "pushes (push_calls=%llu push_calls_with_output="
                   "%llu)\n",a->stall_calls,
                   (unsigned long long)a->push_calls,
                   (unsigned long long)a->push_calls_with_output);
        }
        a->stall_calls=0;
    } else {
        a->stall_calls++;
        if(a->stall_calls==50||(a->stall_calls>50&&a->stall_calls%200==0)){
            /* Fires once at 50 consecutive no-output pushes, then
             * every 200 after that — so a genuinely stuck track logs
             * periodically instead of either going silent forever or
             * flooding the log every single call. This is the direct
             * evidence line to grep for: if this appears and keeps
             * climbing right after a STALE/CLEAN START pair with no
             * matching "recovered after..." line ever following it,
             * that PROVES audio is stuck post-recovery — rather than
             * inferring it from an empty audio track downstream.      */
            printf("[audxc] STALL: %d consecutive pushes with zero "
                   "output (push_calls=%llu push_calls_with_output="
                   "%llu active=%d pts_ok=%d pcm_fill=%d)\n",
                   a->stall_calls,(unsigned long long)a->push_calls,
                   (unsigned long long)a->push_calls_with_output,
                   a->active,a->pts_ok,a->pcm_fill);
        }
    }
    return result;
}

/* ----------------------------------------------------------------
   VIDEO TRANSCODER (optional, only for MPEG-2 source channels)
   H264/MPEG-2 -> libavcodec decode -> libavcodec/libx264 encode
   ---------------------------------------------------------------- */
#define AU_MAX     (1<<20)
#define VOUT_MAX   8192

static enum AVCodecID st2codec(uint8_t st){
    switch(st){case 0x01:case 0x02:return AV_CODEC_ID_MPEG2VIDEO;
    case 0x1B:return AV_CODEC_ID_H264;case 0x24:return AV_CODEC_ID_HEVC;
    default:return AV_CODEC_ID_NONE;}}

typedef struct{uint8_t buf[AU_MAX];int len;int64_t pts,dts;int pts_valid;
    int truncated; /* set by au_push if this AU would have exceeded
        AU_MAX -- previously the overflow was silently truncated and
        the partial AU still got fed to the decoder, which can corrupt
        the decode (a real H264 AU cut off mid-slice-data, not at a
        NAL boundary, can desync the entropy decoder, corrupt SPS/PPS
        state for the picture, or produce random macroblocks). Now:
        the AU is discarded entirely (see vidxc_push's check of this
        flag right before sending to the decoder) and dec_ready is
        reset so the next genuine I-slice re-validates cleanly, same
        recovery path already used for a CC-drop or decode error --
        never partially decode a truncated AU.                        */
}AuBuf;
static void au_reset(AuBuf*a){a->len=0;a->pts_valid=0;a->truncated=0;}

/* Decode the Exp-Golomb slice_type field from a raw (non-emulation-
 * stripped) H264 slice NAL payload, to determine if this is a true
 * I-slice (safe random-access point) rather than just checking for
 * SPS presence elsewhere in the AU. Some real broadcast encoders
 * repeat SPS/PPS on a schedule independent of GOP/IDR boundaries, so
 * "AU contains an SPS NAL" does NOT reliably mean "this AU's picture
 * is safe to decode after a gap" — it can just as easily be an
 * ordinary P-slice that happens to be adjacent to a repeated SPS.
 * `payload` should point just after the NAL header byte; `len` is
 * the number of bytes available (emulation-prevention bytes are
 * handled inline). Returns 1 if this is an I-slice (slice_type 2 or
 * 7), 0 otherwise (including on any parse failure — fail safe).    */
static int is_islice(const uint8_t*payload,int len){
    if(len<=0)return 0;
    /* Strip emulation prevention (00 00 03 -> 00 00) into a small
     * local buffer; slice headers are short, a few dozen bytes is
     * always enough to reach first_mb_in_slice + slice_type.       */
    uint8_t buf[64];int blen=0;
    for(int i=0;i<len&&blen<(int)sizeof(buf);i++){
        if(i+2<len&&payload[i]==0&&payload[i+1]==0&&payload[i+2]==3){
            buf[blen++]=0;if(blen<(int)sizeof(buf))buf[blen++]=0;i+=2;
        }else buf[blen++]=payload[i];
    }
    if(blen<2)return 0;
    uint64_t bitpos=0;
    uint64_t totalbits=(uint64_t)blen*8;
    int err=0;
    int(*read_bit)(uint8_t*,uint64_t*,uint64_t,int*)=NULL;(void)read_bit;
    #define BR_BIT() ({ \
        int _v=0; \
        if(bitpos>=totalbits){err=1;} \
        else{ _v=(buf[bitpos/8]>>(7-(bitpos%8)))&1; bitpos++; } \
        _v; })
    #define BR_UE() ({ \
        int _lz=0; \
        while(!err&&BR_BIT()==0){_lz++;if(_lz>32){err=1;break;}} \
        int _r=0; \
        if(!err){ if(_lz>0){ for(int _k=0;_k<_lz&&!err;_k++)_r=(_r<<1)|BR_BIT(); _r=(1<<_lz)-1+_r; } } \
        _r; })
    (void)BR_UE(); /* first_mb_in_slice — discarded */
    int slice_type=BR_UE();
    #undef BR_BIT
    #undef BR_UE
    if(err)return 0;
    int st=slice_type%5; /* slice_type 5-9 mean "all slices in picture
                             are this type"; mod 5 normalizes 0-4      */
    return st==2; /* 2 == I-slice */
}

/* Push one TS packet into AU accumulator.
 * CRITICAL: discard continuation packets before first PUSI
 * (prevents AVERROR_INVALIDDATA from mid-AU decoder feed).     */
static void au_push(AuBuf*a,const uint8_t*pkt){
    int afc=(pkt[3]>>4)&3;if(afc==2)return;
    int off=4;if(afc==3)off+=1+(int)pkt[4];if(off>=TS_SZ)return;
    int pusi=(pkt[1]>>6)&1;
    const uint8_t*pay=pkt+off;int plen=TS_SZ-off;
    if(pusi){
        if(plen<9||pay[0]!=0||pay[1]!=0||pay[2]!=1)return;
        uint8_t f=(pay[7]>>6)&3;
        if(f&&plen>=14){
            int64_t p=(int64_t)(((uint64_t)(pay[9]&0x0E)<<29)|
                ((uint64_t)pay[10]<<22)|((uint64_t)(pay[11]&0xFE)<<14)|
                ((uint64_t)pay[12]<<7)|((uint64_t)(pay[13]&0xFE)>>1));
            a->pts=p;a->pts_valid=1;
            a->dts=(f==3&&plen>=19)?(int64_t)(((uint64_t)(pay[14]&0x0E)<<29)|
                ((uint64_t)pay[15]<<22)|((uint64_t)(pay[16]&0xFE)<<14)|
                ((uint64_t)pay[17]<<7)|((uint64_t)(pay[18]&0xFE)>>1)):p;
        }
        int skip=9+pay[8];pay+=skip;plen-=skip;
        if(plen<=0){a->len=0;return;}
        a->len=0; /* start fresh AU */
    }else{
        if(a->len==0)return; /* no PUSI yet — discard */
    }
    if(plen>0){
        int cp=plen;
        if(a->len+cp>AU_MAX){
            /* Would overflow -- mark this AU as poisoned. Still cap
             * the copy to whatever room is actually left (never write
             * past the buffer), but the bytes copied past this point
             * are dead data: vidxc_push will discard the whole AU
             * because of the truncated flag, never decode any of it,
             * partial or not. See AuBuf.truncated doc for rationale.  */
            cp=AU_MAX-a->len;
            a->truncated=1;
        }
        if(cp>0){memcpy(a->buf+a->len,pay,cp);a->len+=cp;}
    }
}

typedef struct{
    int             active;
    AVCodecContext *dec_ctx;
    AVPacket       *avpkt;
    AVFrame        *frame;
    AVCodecContext *enc_ctx;
    AVPacket       *enc_pkt;
    AuBuf           au;
    uint8_t        *out;
    int             out_len;
    uint8_t         vid_cc;
    int64_t         out_pts;
    int64_t         last_decoded_pts; /* most recent v->frame->pts seen
        from the decoder - used to detect a decoder that keeps emitting
        SOMETHING (so the existing zero-frame stall check never fires)
        but whose actual content has stopped advancing, a real, distinct
        failure mode confirmed via direct analysis of real production
        segments (see stall_calls doc in vidxc_push for full detail)    */
    int             last_decoded_pts_valid;
    int             stuck_pts_calls; /* consecutive frames decoded with
        an unchanged (or non-advancing) PTS - see vidxc_push           */
    float           crf;
    int             fpn,fpd;
    int             dec_ready;     /* gate: wait for SPS before decoding */
    int             stall_calls;   /* consecutive decode calls with zero
                                       frame output - see the dec_ready
                                       reset logic in vidxc_push for why
                                       this matters and how it's used   */
    int             force_idr;     /* force next encoder output to be a
                                       fresh IDR (used after CC recovery -
                                       see vidxc_push for full rationale) */
    int             got_keyframe;  /* set when encoder outputs an IDR */
    int             enc_started;   /* 1 after first IDR from encoder */
    int             corrupt_errors_since_check; /* incremented by the
        global av_log callback (see install_corruption_log_callback)
        whenever libavcodec logs a real reference-frame corruption
        message ("mmco: unref short failure", "reference picture
        missing during reorder") while THIS context is the one
        actively decoding. Read-and-reset by vidxc_push right after
        each decode call. This is the real, direct signal — not an
        inferred heuristic — for the bug confirmed via real production
        evidence: a single CC drop can leave the decoder in a corrupted
        reference-frame state that does NOT self-heal, with errors
        escalating across many subsequent segments despite zero
        further CC drops, until a fresh I-slice forces re-validation.*/
    int             corrupt_calls;  /* consecutive decode calls (NOT
        consecutive frames - granularity matches stall_calls/
        stuck_pts_calls for consistency) where corrupt_errors_since_check
        was nonzero - see vidxc_push for the actual threshold/reset.  */
    time_t          last_islice_log; /* wall-clock second of the last
        "I-slice found" print -- see that printf's own comment for why
        this exists (rate-limiting a potentially hot-path log line).  */
    /* Full-range -> limited-range color conversion (green/color-shift
     * fix). Some sources signal full-range (0-255, "PC"/JPEG range)
     * YUV via SPS VUI - ffprobe shows this as e.g. "yuvj420p(pc,
     * bt709,...)". The old code did a raw av_frame_copy() straight
     * from that full-range decoded frame into an output frame merely
     * LABELED yuv420p (standard limited/TV range, 16-235), with zero
     * pixel-value rescaling: the exact same numeric YCbCr values then
     * got reinterpreted one range step differently by every
     * downstream player, which is a textbook cause of a greenish/
     * color-shifted picture. range_sws (created lazily, once per
     * resolution) does the real full->limited rescale via sws_scale
     * before encoding, so every player renders correctly regardless
     * of whether it even honors an explicit full-range VUI flag
     * (embedded LG/Samsung decoders in particular are unreliable
     * about that) -- see the source-range check in vidxc_push.        */
    struct SwsContext *range_sws;
    int                range_sws_w,range_sws_h;
    enum AVPixelFormat range_sws_srcfmt;
    /* PTS sanity/wrap-normalize state (see the clamp in vidxc_push
     * right after src_pts is computed from frame->pts, for the full
     * root-cause writeup). Confirmed real, reproducible bug: three
     * separate production segments (6s apart) each showed a decoded
     * frame's pts equal to the CORRECT value plus exactly 8589934592
     * (2^33, the MPEG PTS wrap boundary at 90kHz) — to sub-millisecond
     * precision across all three occurrences — while every other
     * segment's pts was correct. Since au_push's raw 33-bit PTS
     * extraction has no wrap-add logic of its own (confirmed: the
     * only PTS math in this file is a straight bit-field read), this
     * is coming from inside libavcodec's own decode/reorder path.     */
    int64_t last_good_pts;
    int     last_good_pts_valid;
}VidXc;
static VidXc *volatile g_corruption_target=NULL; /* which channel's VidXc
    is "currently decoding" right now, for the log callback below to
    attribute a corruption message to. Set just before
    avcodec_receive_frame, cleared right after — narrow window, correct
    in practice since channel processing is serial within a NIC thread,
    not concurrent (multiple NIC threads for multiple interfaces could
    theoretically race here, but a rare misattribution in that case only
    costs a slightly-delayed recovery trigger on the WRONG channel, not
    a crash or data corruption - an acceptable, bounded risk given the
    alternative is no detection at all).                               */
static void corruption_log_callback(void*avcl,int level,const char*fmt,va_list vl){
    if(level>AV_LOG_ERROR)return; /* only care about real errors, not
        info/debug/warning-level chatter - keeps this cheap on the
        overwhelming majority of log calls during normal operation    */
    char buf[256];
    int n=vsnprintf(buf,sizeof buf,fmt,vl);
    if(n<=0)return;
    /* Original two strings caught a single confirmed real-production
     * incident (post-CC-drop stale reference-frame corruption). This
     * was extended after a side-by-side comparison against plain
     * ffmpeg (-c:v copy) decoding the SAME raw source feed: ffmpeg's
     * demuxer independently flagged genuine "Packet corrupt" /
     * timestamp-discontinuity events on this exact network with no
     * corresponding CC error -- i.e. real payload-level corruption
     * that a copy-mode pipeline never has to decode, but this
     * pipeline's actual H.264 decode step does. That failure mode
     * surfaces as "non-existing PPS/SPS referenced" and "no frame!"
     * from libavcodec -- neither string was in the original match set,
     * so this whole corruption class could hit the decoder and never
     * trigger dec_ready recovery. Confirmed real (not hypothetical):
     * the exact "non-existing PPS 0 referenced" / "no frame!" pattern
     * was captured from this feed via plain ffmpeg in the same window
     * these fixes address.                                            */
    if(strstr(buf,"mmco: unref short failure")||
       strstr(buf,"reference picture missing")||
       strstr(buf,"non-existing PPS")||
       strstr(buf,"non-existing SPS")||
       strstr(buf,"decode_slice_header error")||
       strstr(buf,"no frame!")){
        VidXc*v=g_corruption_target;
        if(v)v->corrupt_errors_since_check++;
    }
    (void)avcl;
}

static int vidxc_init(VidXc*v,uint8_t st,int fpn,int fpd,float crf){
    memset(v,0,sizeof*v);
    enum AVCodecID cid=st2codec(st);
    if(cid==AV_CODEC_ID_NONE){
        fprintf(stderr,"[vidxc] unknown stream_type 0x%02X\n",st);return 0;}
    const AVCodec*dec=avcodec_find_decoder(cid);
    if(!dec){fprintf(stderr,"[vidxc] no decoder\n");return 0;}
    v->dec_ctx=avcodec_alloc_context3(dec);
    v->dec_ctx->flags|=AV_CODEC_FLAG_LOW_DELAY;
    v->dec_ctx->thread_count=1;
    if(avcodec_open2(v->dec_ctx,dec,NULL)<0){
        avcodec_free_context(&v->dec_ctx);return 0;}
    v->avpkt=av_packet_alloc();
    v->frame=av_frame_alloc();
    v->enc_pkt=av_packet_alloc();
    v->out=malloc(VOUT_MAX*TS_SZ);
    v->crf=crf>0?crf:23.0f;
    v->fpn=fpn?fpn:25;v->fpd=fpd?fpd:1;
    au_reset(&v->au); v->active=1;
    printf("[vidxc] decoder: %s fps=%d/%d crf=%.0f\n",dec->name,fpn,fpd,crf);
    return 1;
}
static int vidxc_open_enc(VidXc*v,int w,int h,AVRational sar,
                           enum AVColorSpace cspace,enum AVColorPrimaries cprim,
                           enum AVColorTransferCharacteristic ctrc){
    /* Try encoders in order: OpenH264 -> libx264 -> any H264 */
    const AVCodec*enc=avcodec_find_encoder_by_name("libopenh264");
    if(!enc) enc=avcodec_find_encoder_by_name("libx264");
    if(!enc) enc=avcodec_find_encoder(AV_CODEC_ID_H264);
    if(!enc){fprintf(stderr,"[vidxc] no H264 encoder found\n");return 0;}

    v->enc_ctx=avcodec_alloc_context3(enc);
    v->enc_ctx->width=w; v->enc_ctx->height=h;
    if(sar.num>0&&sar.den>0)v->enc_ctx->sample_aspect_ratio=sar;
    v->enc_ctx->time_base=(AVRational){v->fpd,v->fpn};
    v->enc_ctx->framerate=(AVRational){v->fpn,v->fpd};
    v->enc_ctx->gop_size=v->fpn*2/v->fpd; /* IDR every 2s */
    v->enc_ctx->max_b_frames=0;            /* no B-frames: DTS=PTS */
    v->enc_ctx->pix_fmt=AV_PIX_FMT_YUV420P;
    /* Propagate the SOURCE's real color matrix/primaries/transfer
     * (e.g. bt709 for HD, confirmed via ffprobe on real production
     * streams) so libx264 writes correct VUI instead of leaving it
     * unspecified — an unspecified/wrong matrix is the other classic
     * half of the greenish/color-shifted-picture bug (the range
     * mismatch, fixed in vidxc_push via range_sws, is the other half).
     * color_range is ALWAYS forced to MPEG (limited) here, regardless
     * of the source's range, because vidxc_push always rescales
     * full-range source frames down to limited range before they ever
     * reach the encoder — so limited range is what the encoder is
     * actually receiving, and what must be signalled.
     * NOTE: libopenh264 (tried first, above) does not forward these
     * fields into its output VUI at all (a real ffmpeg-wrapper
     * limitation, not something fixable from this side) — this only
     * takes visible effect on streams that fall back to libx264. The
     * range fix in vidxc_push, however, applies with either encoder.  */
    v->enc_ctx->color_range=AVCOL_RANGE_MPEG;
    v->enc_ctx->colorspace=cspace;
    v->enc_ctx->color_primaries=cprim;
    v->enc_ctx->color_trc=ctrc;
    v->enc_ctx->thread_count=0; /* Single-threaded by design for high
                                    channel-count deployments (e.g. ~300
                                    concurrent streams on shared hardware).
                                    Multithreading (thread_count=0, auto)
                                    was tested and reverted: it raised
                                    per-channel CPU from ~100% to ~150-160%
                                    (real measurement) for a stream that
                                    was already real-time at 100%, and at
                                    high channel counts also risks thread
                                    oversubscription — auto-detect sizes
                                    threads off the WHOLE machine's core
                                    count, not accounting for the other
                                    ~300 channel processes competing for
                                    those same cores, which causes
                                    scheduler thrashing rather than real
                                    speedup. Fixed cost-per-channel matters
                                    more than per-channel speed at this
                                    scale. One packet per frame is also a
                                    nice side benefit (see sliced_threads
                                    below for the setting that actually
                                    matters for that guarantee).
                                    NOTE: this field was left at 0 (auto)
                                    despite this exact comment saying that
                                    was reverted — auto-sizing off the
                                    whole machine's core count with no
                                    awareness of ~300 sibling channel
                                    processes doing the same thing is the
                                    single most likely cause of large
                                    (5-15%, vs an expected 1-3%) run-to-run
                                    CPU variance per channel: which channel
                                    "wins" contested cores at any instant
                                    is scheduler luck, not fixed cost.     */

    if(strcmp(enc->name,"libopenh264")==0){
        /* OpenH264: ABR mode, map CRF quality to bitrate.
         * crf=18~4Mbps, crf=23~2.5Mbps, crf=28~1.5Mbps, crf=32~900kbps.
         * Decay constant verified against this exact ladder: with the
         * old 0.82f, actual output at crf=23/28/32 was only ~1.48/
         * 0.55/0.25 Mbps — 41-72% BELOW the documented/intended
         * bitrate at every CRF except the crf=18 anchor point itself
         * (where the exponent is 0, so no decay constant applies).
         * Every channel at the default crf=23 has been silently
         * encoding at barely half its intended bitrate — the most
         * likely direct cause of poor SD clarity specifically (SD's
         * lower resolution makes under-bitrating far more visible,
         * i.e. blockier/softer, than the same shortfall on HD).
         * 0.90f was solved against all three documented anchor points
         * (18->23, 23->28, 28->32) and matches each within ~7%.        */
        int br=(int)(4000000.0f*powf(0.90f,v->crf-18.0f));
        if(br<400000)br=400000; if(br>8000000)br=8000000;
        v->enc_ctx->bit_rate=br;
        v->enc_ctx->rc_max_rate=br;
        v->enc_ctx->rc_buffer_size=br*2;
        /* Main profile: supported by all LG/Samsung/VLC/ExoPlayer */
        av_opt_set(v->enc_ctx->priv_data,"profile","main",0);
        /* Always allow max quality — no frame skipping */
        av_opt_set_int(v->enc_ctx->priv_data,"allow_skip_frames",0,0);
        /* Complexity: 0=fastest, no B-frames by default in OpenH264 */
        av_opt_set_int(v->enc_ctx->priv_data,"loopfilter",1,0);
    } else if(strcmp(enc->name,"libx264")==0){
        /* Resolution-aware preset: "ultrafast" (the WORST-quality x264
         * preset — minimal motion search, no trellis, no subpel
         * refinement) was hardcoded for every channel regardless of
         * resolution. That's defensible for HD at 300-channel scale
         * (CPU-per-channel matters more there — see the thread_count=0
         * doc above), but SD has vastly fewer macroblocks per frame:
         * a slower/better preset costs proportionally far less CPU on
         * SD than the same preset bump would on HD, so there's no
         * reason to pay full "ultrafast" quality loss on SD too. This
         * is very likely the actual cause of poor SD clarity reported
         * — since this system has no libopenh264 support at all (per
         * the real ffprobe build-config banner, no --enable-libopenh264
         * anywhere), EVERY channel has been running the libx264
         * fallback below, at "ultrafast", the whole time.
         * "veryfast" was chosen for SD (h<=576) as a safe first step:
         * meaningfully better subpel/motion search than ultrafast, at
         * a cost increase that's small in absolute terms specifically
         * because SD has so few macroblocks — bump further (faster/
         * fast/medium) if there's still CPU headroom to spend.        */
        const char*preset=(h<=576)?"veryfast":"ultrafast";
        av_opt_set(v->enc_ctx->priv_data,"preset",preset,0);
        av_opt_set(v->enc_ctx->priv_data,"tune","zerolatency",0);
        /* tune=zerolatency disables CABAC by default, which silently
         * forces x264 into Constrained Baseline profile regardless of
         * any profile setting — re-enable cabac=1 explicitly to get
         * true Main profile output (required by many TV/STB decoders
         * that reject Baseline streams or behave inconsistently with
         * them). ctx->profile is the correct way to request the
         * profile; "profile=main" inside x264-params is NOT a valid
         * key for libx264's param parser and is silently ignored.   */
        v->enc_ctx->profile=FF_PROFILE_H264_MAIN;
        char p[256];
        snprintf(p,sizeof p,
            "crf=%.0f:repeat_headers=1:annexb=1:bframes=0"
            ":sliced_threads=0:cabac=1:aud=1:aq-mode=2:aq-strength=0.8",
            v->crf);
        /* aq-mode=2 (auto-variance, dark-scene biased) + aq-strength=0.8:
         * redistributes bits toward low-variance/flat regions (skin
         * tones, backgrounds, low-motion news/talk content — common on
         * SD broadcast channels) instead of spending them uniformly.
         * This is the standard perceptual-quality lever x264 has that
         * OpenH264 has NO equivalent for at all, so it only helps here
         * because libx264 is what's actually encoding on this build.
         * Cheap: aq analysis, unlike subme/trellis, doesn't meaningfully
         * add CPU cost on top of whatever preset is already chosen —
         * safe to apply at both presets above, not just the SD one.   */
        av_opt_set(v->enc_ctx->priv_data,"x264-params",p,0);
    } else {
        v->enc_ctx->bit_rate=2000000;
    }

    if(avcodec_open2(v->enc_ctx,enc,NULL)<0){
        fprintf(stderr,"[vidxc] encoder open failed (%s)\n",enc->name);
        avcodec_free_context(&v->enc_ctx); return 0;}

    printf("[vidxc] encoder: %s %dx%d %.2ffps  bitrate=%dkbps\n",
           enc->name,w,h,(float)v->fpn/v->fpd,
           (int)(v->enc_ctx->bit_rate/1000));
    return 1;
}
static void vidxc_free(VidXc*v){
    if(!v->active)return;
    avcodec_free_context(&v->enc_ctx);
    avcodec_free_context(&v->dec_ctx);
    av_frame_free(&v->frame);
    av_packet_free(&v->avpkt);
    av_packet_free(&v->enc_pkt);
    if(v->range_sws){sws_freeContext(v->range_sws);v->range_sws=NULL;}
    free(v->out);v->active=0;
}
static void vidxc_write(VidXc*v,const uint8_t*data,int size,
                         uint16_t pid,int64_t pts,int64_t dts,int disc){
    int so=0,first=1;
    while(so<size){
        if(v->out_len+TS_SZ>VOUT_MAX*TS_SZ){
            /* This should not happen with VOUT_MAX sized generously above
             * any realistic frame size, but if it ever does, truncating
             * silently here corrupts the H264 bytestream mid-slice-data
             * (confirmed: this exact path caused real "bytestream
             * overread"/macroblock decode errors on a real broadcast
             * capture when VOUT_MAX was too small). Make it loud instead
             * of silent so a future regression is immediately visible
             * rather than manifesting as mysterious intermittent
             * corruption far away from this code.                       */
            static int _warned=0;
            if(_warned<5){
                fprintf(stderr,"[vidxc] WARNING: frame truncated! "
                        "size=%d exceeds VOUT_MAX=%d packets (%d bytes) — "
                        "increase VOUT_MAX\n",size,VOUT_MAX,VOUT_MAX*TS_SZ);
                _warned++;
            }
            break;
        }
        uint8_t*pkt=v->out+v->out_len;
        memset(pkt,0,TS_SZ); /* zero, not 0xFF — payload area must never
                                 contain stray 0xFF that could be mistaken
                                 for H264 stream data by NAL scanners       */
        pkt[0]=0x47;
        pkt[1]=(uint8_t)((first?0x40:0x00)|((pid>>8)&0x1F));
        pkt[2]=(uint8_t)(pid&0xFF);
        int po;
        if(first){
            /* CRITICAL: carry PCR in the adaptation field of the first TS
             * packet of every video frame. Without PCR the player has no
             * system clock reference and stalls/freezes frame-by-frame
             * even though valid PES data is arriving.                    */
            pkt[3]=(uint8_t)(0x30|(v->vid_cc&0x0F)); /* AFC=11: adapt+payload */
            v->vid_cc=(v->vid_cc+1)&0x0F;
            pkt[4]=7;            /* adaptation_field_length */
            pkt[5]=(uint8_t)(0x10|(disc?0x80:0x00)); /* PCR_flag=1,
                discontinuity_indicator=1 if this frame follows a real
                recovery (see vidxc_write's doc above for full
                rationale — this is the actual fix for VLC-specific
                video freezes on channels that experience CC-drop
                recovery, confirmed to play fine on ExoPlayer/LG with
                the exact same unflagged bytes).                      */
            uint64_t pcr_base=(uint64_t)pts; /* 90kHz base */
            uint16_t pcr_ext=0;  /* 27MHz extension, 0 = aligned to 90kHz */
            pkt[6]=(uint8_t)((pcr_base>>25)&0xFF);
            pkt[7]=(uint8_t)((pcr_base>>17)&0xFF);
            pkt[8]=(uint8_t)((pcr_base>>9)&0xFF);
            pkt[9]=(uint8_t)((pcr_base>>1)&0xFF);
            pkt[10]=(uint8_t)(((pcr_base&1)<<7)|0x7E|((pcr_ext>>8)&1));
            pkt[11]=(uint8_t)(pcr_ext&0xFF);
            int ao=4+1+7; /* TS hdr + adapt_len byte + 7 adapt bytes */
            uint8_t*p=pkt+ao;
            p[0]=0;p[1]=0;p[2]=1;p[3]=0xE0;
            /* PES_packet_length = PES_header_data_length(10) + payload size.
             * Previously hardcoded to 0 — technically a legal "unbounded"
             * marker for video PES per spec, but our packetizer doesn't
             * actually leave the packet unbounded (it's TS-framed normally),
             * and ffmpeg's demuxer choked on the mismatch between the
             * claimed-unbounded length and the actual bounded structure,
             * logging "PES packet size mismatch" and momentarily losing
             * sync — which manifested as spurious "non-existing PPS"
             * warnings on perfectly valid SPS/PPS/IDR data. Real hardware
             * decoders (ExoPlayer, LG, Samsung) are far less forgiving of
             * this than ffmpeg's resync logic, matching the reported
             * macroblock/freeze symptoms. Field is 16 bits; per spec,
             * if the payload would overflow it, fall back to 0 (legal
             * unbounded marker) rather than write a truncated/wrong value.*/
            {
                /* PES_packet_length counts everything AFTER this 16-bit
                 * field itself: 2 flag bytes (p[6],p[7]) + 1 byte for
                 * header_data_length (p[8]) + header_data_length(10) +
                 * the actual payload. That's 2+1+10+size = 13+size.
                 * Previous code wrote 10+size — missing the 3 bytes for
                 * the flags+hdr_len_byte — which meant every single PES
                 * packet's declared length was exactly 3 bytes short of
                 * what was actually delivered. Confirmed by direct
                 * measurement against a real broadcast capture: every
                 * frame's actual payload was exactly claimed_length+3.
                 * This is why ffmpeg logged "Packet corrupt"/"PES packet
                 * size mismatch" on literally every frame in every
                 * segment, and in at least one case caused real MB/MV
                 * concealment errors (visible corruption), not just a
                 * benign warning.                                        */
                int pes_len = 13 + size;
                if(pes_len > 0xFFFF) pes_len = 0; /* spec-legal unbounded */
                p[4]=(uint8_t)((pes_len>>8)&0xFF);
                p[5]=(uint8_t)(pes_len&0xFF);
            }
            p[6]=0x80;p[7]=0xC0;p[8]=0x0A;
            p[9] =(uint8_t)(0x31|((pts>>29)&0x0E));
            p[10]=(uint8_t)((pts>>22)&0xFF);
            p[11]=(uint8_t)(0x01|((pts>>14)&0xFE));
            p[12]=(uint8_t)((pts>>7)&0xFF);
            p[13]=(uint8_t)(0x01|((pts<<1)&0xFE));
            p[14]=(uint8_t)(0x11|((dts>>29)&0x0E));
            p[15]=(uint8_t)((dts>>22)&0xFF);
            p[16]=(uint8_t)(0x01|((dts>>14)&0xFE));
            p[17]=(uint8_t)((dts>>7)&0xFF);
            p[18]=(uint8_t)(0x01|((dts<<1)&0xFE));
            po=ao+19; first=0;
        } else {
            po=4;
            int remain=size-so;
            if(remain<TS_SZ-4){
                /* LAST packet of this frame and it won't fill the TS
                 * packet exactly. Use a proper adaptation field with
                 * stuffing_byte padding (AFC=11) — this is the ONLY
                 * spec-legal way to pad a TS packet. Writing raw 0xFF
                 * directly into the payload area (old bug) corrupts the
                 * H264 byte stream: NAL scanners in hardware decoders
                 * (Android/LG/Samsung) don't respect PES packet_length
                 * and parse the stuffing as bogus trailing NAL/start-code
                 * bytes, producing macroblock corruption on every frame
                 * and confusing frame boundaries enough to stall decode
                 * pacing (the "frame by frame" freeze).                  */
                int pad=(TS_SZ-4)-remain; /* stuffing bytes needed */
                pkt[3]=(uint8_t)(0x30|(v->vid_cc&0x0F)); /* AFC=11 */
                v->vid_cc=(v->vid_cc+1)&0x0F;
                int afl=pad-1; /* adaptation_field_length itself counts
                                  as 1 of the pad bytes via its own field */
                if(afl<0)afl=0;
                pkt[4]=(uint8_t)afl;
                if(afl>0){
                    pkt[5]=0x00; /* no flags set, pure stuffing follows */
                    for(int i=1;i<afl;i++)pkt[5+i]=0xFF; /* stuffing_byte */
                }
                po=4+1+afl; /* TS hdr + adapt_len byte + adapt field body */
            } else {
                pkt[3]=(uint8_t)(0x10|(v->vid_cc&0x0F)); /* AFC=01: full payload */
                v->vid_cc=(v->vid_cc+1)&0x0F;
            }
        }
        int sp=TS_SZ-po,cp=size-so;if(cp>sp)cp=sp;
        memcpy(pkt+po,data+so,cp);so+=cp;v->out_len+=TS_SZ;
    }
}
/* Returns number of output TS packets produced */
static int vidxc_push(VidXc*v,const uint8_t*ts_pkt,uint16_t pid){
    v->out_len=0;
    int pusi=(ts_pkt[1]>>6)&1;
    if(pusi&&v->au.len>0){
        /* A truncated AU (see AuBuf.truncated doc) must never reach
         * the decoder, partial or not -- a real H264 AU cut off
         * mid-slice-data (not at a NAL boundary) can desync the
         * entropy decoder or corrupt SPS/PPS state for the picture.
         * Discard it entirely and force I-slice re-validation on the
         * next AU, exactly like the existing CC-drop/decode-error
         * recovery path already does.                                 */
        if(v->au.truncated){
            static int _trunc_warn=0;
            if(_trunc_warn++<5)
                fprintf(stderr,"[vidxc] WARNING: discarding truncated AU "
                        "(%d bytes, exceeded AU_MAX) -- forcing I-slice "
                        "re-validation\n",v->au.len);
            v->dec_ready=0;au_reset(&v->au);
            v->stuck_pts_calls=0;v->last_decoded_pts_valid=0;
            goto push_done;
        }
        /* Gate: only send AU to decoder once we've seen SPS (NAL 7).
         * Without this, decoder gets P-frames first -> INVALIDDATA.  */
        if(!v->dec_ready){
            const uint8_t*d=v->au.buf;int sz=v->au.len;
            int found_islice=0;
            for(int i=0;i+3<sz;i++){
                int nt=-1;int hdr_off=-1;
                if(d[i]==0&&d[i+1]==0&&d[i+2]==0&&i+4<sz&&d[i+3]==1){nt=d[i+4]&0x1F;hdr_off=i+4;}
                else if(d[i]==0&&d[i+1]==0&&d[i+2]==1){nt=d[i+3]&0x1F;hdr_off=i+3;}
                else continue;
                if(nt==1||nt==5){
                    /* found a slice NAL — check if it's really an
                     * I-slice, regardless of whether nt==5 (IDR) or
                     * nt==1 (this source uses non-IDR NALs for what
                     * are functionally I-slices, a known real quirk
                     * of this broadcast encoder — see is_islice doc) */
                    int payload_off=hdr_off+1;
                    if(is_islice(d+payload_off,sz-payload_off)){found_islice=1;}
                    break; /* only the first slice NAL matters — an AU
                              has one picture, one slice_type for our
                              purposes (multi-slice pictures all share
                              the same type in practice for this gate) */
                }
            }
            if(!found_islice){au_reset(&v->au);goto push_done;}
            /* Found a genuine I-slice — flush decoder of any stale
             * reference frames, then let this AU through.
             * Rate-limited: this fires every time dec_ready transitions
             * 0->1, which on a noisy source (frequent CC drops, or
             * frequent truncated-AU resyncs) could otherwise print many
             * times per second per channel -- a real, measurable cost
             * at high channel counts (stdout lock contention, syscall
             * overhead). Cap at once per second per channel; the
             * SIGUSR1 dump and stats line already give visibility into
             * dec_ready's current state without needing every single
             * transition logged.                                       */
            avcodec_flush_buffers(v->dec_ctx);
            v->dec_ready=1;
            v->stuck_pts_calls=0;v->last_decoded_pts_valid=0;
            v->last_good_pts_valid=0; /* fresh decode restart — don't
                let the wrap/sanity clamp compare against a pts from
                before this discontinuity.                            */
            {time_t now_t=time(NULL);
             if(now_t!=v->last_islice_log){
                 printf("[vidxc] I-slice found — decoder flushed and ready\n");
                 v->last_islice_log=now_t;}}
        }
        v->avpkt->data=v->au.buf;v->avpkt->size=v->au.len;
        v->avpkt->pts=v->au.pts_valid?v->au.pts:AV_NOPTS_VALUE;
        v->avpkt->dts=v->au.pts_valid?v->au.dts:AV_NOPTS_VALUE;
        {
            g_corruption_target=v; /* see corruption_log_callback —
                narrow window around the actual decode work so any
                corruption message gets attributed to this channel    */
            int _sp=avcodec_send_packet(v->dec_ctx,v->avpkt);
            static int _dec_log=0;
            if(!_dec_log&&_sp==0){printf("[vidxc] first send_packet OK len=%d\n",v->avpkt->size);_dec_log=1;}
            if(_sp!=0){
                static int _se=0;if(_se++<3)printf("[vidxc] send_packet err=%d len=%d\n",_sp,v->avpkt->size);
                /* CRITICAL: a decode failure must not be silently
                 * tolerated forever. dec_ready only gets reset to 0 on
                 * a CC drop — without this, a single bad AU right
                 * after a recovery (plausible: CC drops mean lost
                 * packets, and the AU that triggered dec_ready=1 can
                 * itself still be missing trailing data from that same
                 * loss event) would leave dec_ready stuck at 1
                 * forever, permanently skipping I-slice re-validation
                 * for every subsequent AU — even perfectly healthy
                 * ones — with no path back to a working decoder state
                 * until another CC drop happens to occur. Confirmed
                 * directly against two real production logs: exactly
                 * this sequence (CC drop -> one decode attempt -> dead
                 * silence) left the segment-cutting pipeline frozen
                 * (segs= never incremented again) for 30+ minutes of
                 * otherwise-healthy packet flow, in both cases.        */
                v->dec_ready=0;au_reset(&v->au);
                v->stuck_pts_calls=0;v->last_decoded_pts_valid=0;
                goto push_done;
            }
            static int _frm=0;
            int _frames_this_call=0;
            if(_sp==0){
            while(avcodec_receive_frame(v->dec_ctx,v->frame)==0){
                _frames_this_call++;
                _frm++;if(_frm<=3)printf("[vidxc] frame%d decoded %dx%d\n",_frm,v->frame->width,v->frame->height);
                int w=v->frame->width,h=v->frame->height;
                AVRational src_sar=v->frame->sample_aspect_ratio;
                if(!v->enc_ctx){
                    if(!w||!h){av_frame_unref(v->frame);continue;}
                    if(!vidxc_open_enc(v,w,h,src_sar,
                                       (enum AVColorSpace)v->frame->colorspace,
                                       (enum AVColorPrimaries)v->frame->color_primaries,
                                       (enum AVColorTransferCharacteristic)v->frame->color_trc)){
                        av_frame_unref(v->frame);continue;}
                }
                AVFrame*ef=av_frame_alloc();
                ef->format=AV_PIX_FMT_YUV420P;
                ef->width=w;ef->height=h;
                ef->sample_aspect_ratio=src_sar;
                ef->color_range=AVCOL_RANGE_MPEG; /* always — see full-
                    range conversion below and the doc on enc_ctx's
                    color_range assignment in vidxc_open_enc.          */
                ef->colorspace=v->frame->colorspace;
                ef->color_primaries=v->frame->color_primaries;
                ef->color_trc=v->frame->color_trc;
                av_frame_get_buffer(ef,0);
                /* Full-range -> limited-range fix (see VidXc.range_sws
                 * doc for the full root-cause explanation). Detect a
                 * full-range ("PC"/JPEG) source either via the
                 * yuvj4:2:0-family pixel formats (the classic signal —
                 * ffprobe shows this as "yuvj420p(pc,...)") or via an
                 * explicit AVCOL_RANGE_JPEG tag on frames already in a
                 * plain yuv420p format. Only THIS path needs an actual
                 * sws_scale rescale of pixel values; ordinary
                 * already-limited-range sources keep the cheap
                 * av_frame_copy() with zero extra cost or behavior
                 * change.                                              */
                enum AVPixelFormat sfmt=(enum AVPixelFormat)v->frame->format;
                int full_range=(sfmt==AV_PIX_FMT_YUVJ420P||sfmt==AV_PIX_FMT_YUVJ422P||
                                sfmt==AV_PIX_FMT_YUVJ444P||sfmt==AV_PIX_FMT_YUVJ440P||
                                v->frame->color_range==AVCOL_RANGE_JPEG);
                if(!full_range){
                    av_frame_copy(ef,v->frame);
                }else{
                    if(!v->range_sws||v->range_sws_w!=w||v->range_sws_h!=h||
                       v->range_sws_srcfmt!=sfmt){
                        if(v->range_sws){sws_freeContext(v->range_sws);v->range_sws=NULL;}
                        v->range_sws=sws_getContext(w,h,sfmt,w,h,AV_PIX_FMT_YUV420P,
                                                     SWS_POINT,NULL,NULL,NULL);
                        v->range_sws_w=w;v->range_sws_h=h;v->range_sws_srcfmt=sfmt;
                        if(v->range_sws){
                            int brightness,contrast,saturation,srcRange,dstRange;
                            const int *inv_tbl,*fwd_tbl;
                            sws_getColorspaceDetails(v->range_sws,(int**)&inv_tbl,
                                &srcRange,(int**)&fwd_tbl,&dstRange,
                                &brightness,&contrast,&saturation);
                            const int*coeff=sws_getCoefficients(
                                v->frame->colorspace==AVCOL_SPC_BT709?
                                SWS_CS_ITU709:SWS_CS_ITU601);
                            /* srcRange=1 (full/PC) -> dstRange=0 (limited/TV):
                             * this is the actual value rescale (0-255 -> 16-235
                             * luma, 1-254 -> 16-240 chroma) that the old
                             * av_frame_copy() never did.                      */
                            sws_setColorspaceDetails(v->range_sws,coeff,1,
                                                      coeff,0,brightness,
                                                      contrast,saturation);
                        }
                    }
                    if(v->range_sws)
                        sws_scale(v->range_sws,(const uint8_t*const*)v->frame->data,
                                  v->frame->linesize,0,h,ef->data,ef->linesize);
                    else
                        av_frame_copy(ef,v->frame); /* sws alloc failed —
                            fall back to old (mistagged but non-crashing)
                            behavior rather than drop the frame.         */
                }
                /* Deinterlace: proper field-blend on all planes (Y, Cb, Cr).
                 * For interlaced source (1080i), blend odd lines with
                 * their neighbors to create progressive output.
                 * This eliminates combing artifacts on LG/Samsung.   */
                int interlaced=0;
#ifdef AV_FRAME_FLAG_INTERLACED
                interlaced=(v->frame->flags&AV_FRAME_FLAG_INTERLACED)!=0;
#else
                interlaced=v->frame->interlaced_frame;
#endif
                if(interlaced){
                    /* Process all 3 planes: Y (full res), Cb, Cr (half res) */
                    int plane_h[3]={h,h/2,h/2};
                    int plane_w[3]={w,w/2,w/2};
                    for(int pl=0;pl<3;pl++){
                        int ph=plane_h[pl],pw=plane_w[pl];
                        for(int y=1;y<ph-1;y+=2){
                            uint8_t*prev=ef->data[pl]+(y-1)*ef->linesize[pl];
                            uint8_t*cur =ef->data[pl]+y    *ef->linesize[pl];
                            uint8_t*next=ef->data[pl]+(y+1)*ef->linesize[pl];
                            for(int x=0;x<pw;x++)
                                cur[x]=(uint8_t)((prev[x]+next[x])>>1);
                        }
                    }
                }
                /* Use the DECODER's own output PTS for A/V sync, not the
                 * PTS of whatever AU was most recently fed in. This source
                 * has B-frames (decode order != presentation order), and
                 * libavcodec's H264 decoder already reorders frames into
                 * correct presentation order internally — frame->pts
                 * reflects that reordering correctly. v->au.pts is the
                 * PTS of the AU most recently PUSHED to the decoder
                 * (submission/decode order), which is NOT the same frame
                 * as the one just RECEIVED when B-frames are buffered
                 * inside the decoder's reorder queue. Using au.pts here
                 * silently stamped every output frame with a stale/wrong
                 * timestamp, producing a wildly out-of-order PTS sequence
                 * in the final TS (confirmed: 211522, 215122, 222322,
                 * 240322, 233122, 229522... instead of monotonically
                 * increasing) — exactly matching "plays like slow motion
                 * frame by frame in VLC" (VLC has to wait/seek across out
                 * of order timestamps), "macroblocks in LG" (a hardware
                 * decoder fed frames in the wrong presentation order),
                 * and "no playback in ExoPlayer" (strict players reject
                 * non-monotonic timestamps outright). Verified via a
                 * direct test harness using this exact AU-build/decode
                 * path: frame->pts came back perfectly monotonic
                 * (209722, 213322, 216922, 220522...) every single time. */
                int64_t src_pts=v->frame->pts!=AV_NOPTS_VALUE?v->frame->pts:AV_NOPTS_VALUE;
                /* PTS wrap-normalize / sanity clamp — fixes a confirmed,
                 * exactly-reproduced bug: real production segments (169-
                 * 177, 6s apart) showed frame->pts periodically equal to
                 * the correct value plus EXACTLY 8589934592 (2^33 ticks,
                 * the standard 90kHz MPEG PTS wrap boundary ˜26.5h) — to
                 * sub-millisecond precision, three separate times, with
                 * every other segment correct. That magnitude of jump
                 * (from ~141s to ~95585s) makes the container's reported
                 * duration explode to ~1105s / ~26h for what's actually
                 * a normal 2s segment, which is exactly consistent with
                 * both reported symptoms: players either can't reconcile
                 * such an impossible timeline and drop/stutter the video,
                 * or reject the audio track outright as unplayable given
                 * how far its (correctly-timed, separately-confirmed)
                 * pts now sits from this exploded video timeline.
                 * au_push's raw PTS extraction has no wrap-add logic at
                 * all (verified — it's a plain 33-bit bitfield read), so
                 * this is originating inside libavcodec's own decode/
                 * reorder path, not this code's PES parsing. Rather than
                 * chase the exact internal libavcodec mechanism, this
                 * clamp is a direct, verifiable fix for the confirmed
                 * symptom: undo an apparent ±2^33 shift when doing so
                 * makes the delta from the last good frame sane again,
                 * and as a final backstop, refuse to pass through any
                 * frame-to-frame jump bigger than 10s of 90kHz ticks
                 * (900000) — no real GOP/segment boundary ever legitimately
                 * jumps a single decoded frame that far — holding at the
                 * last good pts plus one nominal frame duration instead.  */
                if(src_pts!=AV_NOPTS_VALUE){
                    int _was_anchored=v->last_good_pts_valid;
                    if(v->last_good_pts_valid){
                        static const int64_t WRAP33=8589934592LL; /* 2^33 */
                        int64_t delta=src_pts-v->last_good_pts;
                        if(delta>(WRAP33/2))src_pts-=WRAP33;
                        else if(delta<-(WRAP33/2))src_pts+=WRAP33;
                        delta=src_pts-v->last_good_pts;
                        if(delta<0||delta>900000){
                            src_pts=v->last_good_pts+
                                (int64_t)90000*v->fpd/v->fpn;
                        }
                    }
                    v->last_good_pts=src_pts;v->last_good_pts_valid=1;
                    if(!_was_anchored){
                        /* Diagnostic: raw video PTS anchor, to compare
                         * against "audio PTS anchor" in pkt_process for
                         * the same time window — chasing a confirmed
                         * real bug where video and audio ended up on
                         * absolute PTS epochs ~2 hours apart despite
                         * both allegedly reading real PES PTS off the
                         * same source. Fires once per fresh decode
                         * restart (dec_ready 0->1), not once per
                         * channel lifetime, since last_good_pts_valid
                         * is cleared on every I-slice re-validation —
                         * that's fine/expected for this comparison,
                         * just match timestamps against the nearest
                         * audio anchor log line when reading this.      */
                        printf("[vidxc] video PTS anchor: src_pts=%lld "
                               "(%.3fs)\n",(long long)src_pts,
                               (double)src_pts/90000.0);
                    }
                }
                /* Stuck-PTS stall detection: DISABLED.
                 *
                 * This was added to catch a decoder stuck on stale/
                 * corrupted reference frames (real evidence: 3
                 * consecutive production segments with video PCR
                 * frozen at an identical value, alongside genuine
                 * H264 decode errors). Two attempts at this heuristic
                 * (first using <= for "non-increasing", then tightened
                 * to == for "exact repeat") were each followed by real,
                 * reported regressions — the second specifically
                 * causing VLC to freeze within 5 seconds, reliably,
                 * which could not be reproduced in testing. Since this
                 * logic writes to v->dec_ready (the same gate
                 * controlling whether ANY video frame passes through),
                 * a false positive here can fully block video output,
                 * not just degrade it — too severe a risk to keep
                 * active without being able to verify it's safe
                 * against real traffic in isolation first. The
                 * original bug this targeted is real but rare, and
                 * per the same real evidence that found it, appears to
                 * self-recover within a handful of segments even
                 * without this check — a missed catch costs a brief
                 * stutter; an unverified heuristic here has twice cost
                 * much worse. src_pts is still computed above (needed
                 * for the encoder's real PTS feed below) — only the
                 * stall-tracking logic itself is removed.             */
                /* Feed the encoder the REAL source pts (not an arbitrary
                 * counter) so that, even with real multithreading enabled
                 * (frame-level parallelism can buffer/delay frames
                 * internally — see ffmpeg's own threading docs), we can
                 * correctly recover which source frame a given output
                 * packet corresponds to by reading v->enc_pkt->pts back,
                 * rather than assuming output-call-order matches
                 * input-call-order (that assumption silently breaks under
                 * frame-threading and would reproduce the same class of
                 * out-of-order-timestamp bug already found and fixed on
                 * the decoder side earlier).                            */
                ef->pts=(src_pts!=AV_NOPTS_VALUE)?src_pts:(v->out_pts*90000LL*v->fpd/v->fpn);
                int this_frame_is_recovery=v->force_idr; /* capture before
                    it gets cleared below — this is the actual signal
                    for whether the resulting output packet's PCR
                    represents a real discontinuity (see vidxc_write
                    call further down, and its own doc, for why this
                    matters: VLC needs discontinuity_indicator set on
                    exactly this frame, ExoPlayer/LG tolerate it either
                    way).                                               */
                if(v->force_idr){
                    ef->pict_type=AV_PICTURE_TYPE_I;
#ifdef AV_FRAME_FLAG_KEY
                    ef->flags|=AV_FRAME_FLAG_KEY;
#else
                    ef->key_frame=1;
#endif
                    /* NOTE: deliberately NOT calling
                     * avcodec_flush_buffers(v->enc_ctx) here. That call
                     * is documented to interact badly with libx264's
                     * internal lookahead thread — confirmed via direct
                     * testing: it reproduces "lookahead thread is
                     * already stopped" warnings (483 in one 90s test
                     * run alone), recurring on every CC-recovery event.
                     * Forcing pict_type=I above already achieves what
                     * we actually need — x264 won't reference any
                     * prior (pre-gap) frame when encoding a true
                     * I-frame, regardless of internal buffer state —
                     * without touching the encoder's thread machinery.*/
                    v->force_idr=0;
                }
                if(avcodec_send_frame(v->enc_ctx,ef)==0){
                    uint8_t*combined=NULL;int clen=0;
                    int is_key=0;
                    int64_t out_pts=AV_NOPTS_VALUE;
                    while(avcodec_receive_packet(v->enc_ctx,v->enc_pkt)==0){
                        if(v->enc_pkt->flags&AV_PKT_FLAG_KEY)is_key=1;
                        out_pts=v->enc_pkt->pts; /* the pts THIS packet's
                                                     source frame actually
                                                     carried, correctly
                                                     tracked by libavcodec
                                                     across any internal
                                                     buffering delay      */
                        combined=realloc(combined,clen+v->enc_pkt->size);
                        memcpy(combined+clen,v->enc_pkt->data,v->enc_pkt->size);
                        clen+=v->enc_pkt->size;
                        av_packet_unref(v->enc_pkt);
                    }
                    if(combined&&clen>0){
                        src_pts=(out_pts!=AV_NOPTS_VALUE)?out_pts:src_pts;
                        /* Strip SEI NALs (type 6) — see comment above this
                         * block in vidxc_push for full rationale. libx264
                         * emits one of these (its own version string) on
                         * the first IDR; it triggers a known ExoPlayer/
                         * Media3 SeiReader/ReorderingBufferQueue crash on
                         * load. We don't need this NAL for anything, so
                         * just remove it from our own output entirely.   */
                        uint8_t*filtered=malloc(clen);
                        int flen=0;
                        int ci=0;
                        while(ci<clen){
                            int start_code_len=0;
                            if(ci+3<clen&&combined[ci]==0&&combined[ci+1]==0&&combined[ci+2]==1)
                                start_code_len=3;
                            else if(ci+4<clen&&combined[ci]==0&&combined[ci+1]==0&&
                                    combined[ci+2]==0&&combined[ci+3]==1)
                                start_code_len=4;
                            if(start_code_len==0){
                                /* not at a start code (shouldn't happen at
                                 * ci==0, but guard anyway) — copy single
                                 * byte and advance to stay safe           */
                                filtered[flen++]=combined[ci++];
                                continue;
                            }
                            int nal_start=ci;
                            int nal_hdr=ci+start_code_len;
                            int nal_type=(nal_hdr<clen)?(combined[nal_hdr]&0x1F):0;
                            /* find next start code (or end of buffer) to
                             * know where this NAL ends                   */
                            int next=nal_hdr;
                            while(next+2<clen){
                                if(combined[next]==0&&combined[next+1]==0&&
                                   (combined[next+2]==1||
                                    (next+3<clen&&combined[next+2]==0&&combined[next+3]==1)))
                                    break;
                                next++;
                            }
                            int nal_end=(next+2<clen)?next:clen;
                            if(nal_type!=6){
                                memcpy(filtered+flen,combined+nal_start,nal_end-nal_start);
                                flen+=nal_end-nal_start;
                            }
                            ci=nal_end;
                        }
                        if(is_key)v->got_keyframe=1;
                        int64_t vpts=src_pts!=AV_NOPTS_VALUE?src_pts:v->out_pts*90000LL*v->fpd/v->fpn;
                        /* Write every frame the encoder produces — including
                         * the very first one. This flag used to gate on
                         * v->enc_started, but that flag is only set by the
                         * CALLER (pkt_process), and only AFTER this function
                         * (vidxc_push) returns. On the very first IDR, that
                         * means enc_started is still 0 at this exact point
                         * — so the first frame (the ONLY one carrying the
                         * encoder's SPS/PPS, since repeat_headers=1 only
                         * repeats them on IDRs, and this IS the first IDR)
                         * was silently dropped here and never written to
                         * v->out at all. Every later IDR worked fine
                         * because by then enc_started was already 1 from
                         * a previous call. Confirmed via real broadcast
                         * capture: segment 0 (the very first segment) had
                         * zero SPS/PPS NALs in its entire video payload —
                         * 0 of 49 frames decoded — while segments 1-3
                         * decoded perfectly. This exactly matches "video
                         * freezes/macroblocks from the start, ExoPlayer
                         * refuses playback" — a player loading segment 0
                         * first hits a stream with no parameter sets at
                         * all. is_key/got_keyframe is still used by the
                         * caller to decide whether THIS frame should
                         * trigger seg_open()/segment-cut; that logic is
                         * unaffected by removing the write-side gate here. */
                        vidxc_write(v,filtered,flen,pid,vpts,vpts,this_frame_is_recovery);
                        free(filtered);
                        free(combined);
                    }
                }
                v->out_pts++;
                av_frame_free(&ef);av_frame_unref(v->frame);
            }
            if(_frames_this_call>0){
                v->stall_calls=0;
            } else {
                /* send_packet succeeded but produced no frame this
                 * call. This is NORMAL and expected on sources using
                 * B-frames (confirmed this source does: direct check
                 * found B/I/P all present) — the decoder legitimately
                 * buffers a few frames internally for reorder before
                 * its first output. Only treat this as a genuine stall
                 * (and reset dec_ready to force I-slice re-validation)
                 * after a SUSTAINED run with zero output — long enough
                 * to rule out normal reorder delay (bounded to a
                 * handful of frames, never dozens), but short enough
                 * to recover quickly from a real post-CC-drop decode
                 * stall (the actual bug this is fixing — see the
                 * send_packet-error branch above for full rationale,
                 * confirmed against two real production logs where
                 * this exact failure mode froze segment output for
                 * 30+ minutes with no recovery).                      */
                v->stall_calls++;
                if(v->stall_calls>=30){
                    v->dec_ready=0;v->stall_calls=0;
                    v->stuck_pts_calls=0;v->last_decoded_pts_valid=0;
                }
            }
            /* Real, sustained reference-frame corruption check — runs
             * UNCONDITIONALLY regardless of whether frames were
             * produced this call, since the confirmed real bug shows
             * frames DO keep being produced while corrupted (that is
             * exactly why the zero-frame stall check above never
             * caught this in the first place). Directly observed via
             * libavcodec's own log output (corrupt_errors_since_check,
             * fed by corruption_log_callback) — not an inferred
             * heuristic like the disabled stuck-PTS check. Confirmed
             * against real production evidence: a single CC drop left
             * "mmco: unref short failure"/"reference picture missing"
             * errors escalating across 6+ consecutive segments with
             * zero further CC drops to ever trigger recovery any other
             * way. Also covers "non-existing PPS/SPS referenced" and
             * "no frame!" — the failure signature of genuine payload
             * corruption (confirmed via a side-by-side ffmpeg -c:v copy
             * run on the same feed logging independent "Packet corrupt"
             * demuxer events with no corresponding CC error) reaching
             * the decoder without ever tripping a CC-based check. See
             * corruption_log_callback for the full current string set.
             * Threshold of 5 consecutive calls (much tighter than
             * stall_calls' 30) is appropriate here because this is a
             * direct signal — a real corruption error logged
             * repeatedly is unambiguous, unlike "zero frames" which
             * has a legitimate explanation in normal B-frame reorder
             * delay.                                                  */
            if(v->corrupt_errors_since_check>0){
                v->corrupt_calls++;
                if(v->corrupt_calls>=5){
                    v->dec_ready=0;v->corrupt_calls=0;
                    v->stall_calls=0;v->stuck_pts_calls=0;
                    v->last_decoded_pts_valid=0;
                }
            } else {
                v->corrupt_calls=0;
            }
            v->corrupt_errors_since_check=0;
            g_corruption_target=NULL;
            }}
        au_reset(&v->au);
    }
push_done:
    au_push(&v->au,ts_pkt);
    return v->out_len/TS_SZ;
}

/* ----------------------------------------------------------------
   AES-128-CBC
   ---------------------------------------------------------------- */
typedef struct{AES_KEY enc;uint8_t iv0[AES_BLK],iv[AES_BLK];
               uint8_t buf[WRITE_BATCH*TS_SZ+AES_BLK*2];int n;}Aes;
static void aes_init(Aes*a,const uint8_t*k,const uint8_t*iv){
    AES_set_encrypt_key(k,128,&a->enc);
    memcpy(a->iv0,iv,AES_BLK);memcpy(a->iv,iv,AES_BLK);a->n=0;}
static void aes_reset(Aes*a){memcpy(a->iv,a->iv0,AES_BLK);a->n=0;}
static void aes_flush(Aes*a,int fd,int fin){
    int n=a->n;
    if(fin){int p=AES_BLK-(n%AES_BLK);if(!p)p=AES_BLK;memset(a->buf+n,p,p);n+=p;}
    else n=(n/AES_BLK)*AES_BLK;
    if(n<=0)return;
    uint8_t out[sizeof(a->buf)];
    AES_cbc_encrypt(a->buf,out,n,&a->enc,a->iv,AES_ENCRYPT);
    ssize_t r=write(fd,out,n);(void)r;
    int rem=fin?0:a->n-n;
    if(rem>0)memmove(a->buf,a->buf+n,rem);a->n=rem;}
static void aes_push(Aes*a,int fd,const uint8_t*p){
    memcpy(a->buf+a->n,p,TS_SZ);a->n+=TS_SZ;
    if(a->n>=WRITE_BATCH*TS_SZ)aes_flush(a,fd,0);}

/* ----------------------------------------------------------------
   CHANNEL OPTIONS & STATE
   ---------------------------------------------------------------- */
typedef struct{
    int   audio_transcode,video_transcode;
    int   fix_mp2,fix_interlace,pts_reset;
    float video_crf;
    int   audio_bitrate;   /* AAC target bitrate, default 128000        */
    int   audio_map;       /* 1-indexed position of a SINGLE audio track
                               to select (by PMT order); 0 = not set,
                               meaning select ALL discovered audio
                               tracks instead (default multi-track
                               behavior). When set, every OTHER audio
                               track is dropped entirely from output.   */
    char  audio_coder[16]; /* AAC encoder algorithm: "twoloop" (default,
                               best quality/bit) or "fast" (measured
                               ~2.4x lower CPU, slightly less optimal
                               bit allocation at the same bitrate — see
                               audxc_init for benchmark numbers). Native
                               ffmpeg AAC encoder only; libfdk_aac is not
                               available in this build environment and
                               aac_mode=vbrN is an libfdk_aac-specific
                               option name that the native encoder does
                               not implement (native AAC is CBR-only,
                               controlled via bit_rate).                */
    int   simple_cut;      /* 0 (default) = has_keyframe: Exp-Golomb
                               slice_type check, correctly recognizes
                               non-standard nal_type=1 functional
                               keyframes on sources that need it. 1 =
                               scan_nal-only: raw NAL type check only
                               (matches the older v23 codebase's
                               cut logic), no Exp-Golomb parsing. Only
                               affects the simple-copy/audio_transcode
                               segment-cut decision — video_transcode's
                               internal decoder gate (vidxc_push) always
                               uses the full has_keyframe/is_islice
                               check regardless of this setting, since
                               it has a stricter requirement (decoder
                               needs a REAL safe-to-decode point, not
                               just a reasonable place to cut a file)
                               and switching that gate risks the decoder
                               never starting at all on sources that
                               only use nal_type=1, not real IDRs.      */
    int   idr_rewrite;     /* 0 (default) = off, leave the cut-point
                               slice NAL exactly as the source sent it.
                               1 = rewrite the cut-point slice's NAL
                               type from 1 to 5 (real IDR) in the
                               simple-copy/audio_transcode segment-cut
                               path only, on the EXACT packet identified
                               as a cut point (never elsewhere in the
                               stream). See rewrite_slice_to_idr() for
                               full rationale: this source's encoder
                               never emits real type-5 IDRs (confirmed:
                               zero across 8 real consecutive captured
                               segments), only Exp-Golomb-confirmed
                               all-intra type-1 slices, which leaves
                               every segment boundary structurally
                               unable to tell a decoder to discard
                               prior reference-buffer state — directly
                               confirmed as the cause of "mmco: unref
                               short failure"/"reference picture
                               missing" errors on every one of those 8
                               segments. Off by default until verified
                               against a real player on this source,
                               since the rewrite has one acknowledged,
                               untested gap (frame_num/POC continuity
                               at the rewritten boundary — see the
                               function doc for detail).                */
}ChOpts;

/* ----------------------------------------------------------------
   v23-COMPATIBLE SIMPLE CUT (simple_cut=1)
   ---------------------------------------------------------------- */
/* Implemented inline at each call site via nal_type(p)==5||==20 —
 * matches v23's scan_nal()'s IDR detection (raw NAL type only, no
 * Exp-Golomb slice_type parsing). See simple_cut option doc above
 * for full rationale.                                               */

/* Multi-audio-track support: each selected/transcoded source audio
 * track gets its own slot. src_pid is the PID as it appears in the
 * SOURCE PMT; out_pid is the PID we actually write transcoded or
 * passed-through output on. For the first/only track in the common
 * single-track case, out_pid==src_pid (reusing the original PID for
 * a 1:1 replace/passthrough is simplest and most compatible) — only
 * additional tracks beyond the first need a distinct, synthesized
 * output PID, since two different streams can't share one PID.      */
typedef struct{
    uint16_t src_pid,out_pid;
    uint8_t  stream_type;
    AudXc    audxc;
    uint8_t  cc;
    uint64_t write_loop_entries; /* number of times the audio dispatch
        code's pkt_write loop body actually executed (i.e. nout>0 was
        true and at least one TS-packet-sized chunk was written) -
        for the SIGUSR1 dump, to pin down precisely where real,
        confirmed-produced AAC output (see audxc.push_calls_with_
        output) might be getting lost before it reaches the segment
        file. Confirmed real gap: a production dump showed
        push_calls_with_output climbing (real output being produced)
        at the exact same moment 7 consecutive real segments showed
        zero audio packets — this counter narrows down whether the
        write loop itself is even running.                            */
    uint64_t write_loop_packets; /* total TS-packet-sized chunks
        actually passed to pkt_write across all write_loop_entries -
        lets the dump distinguish "loop ran once but wrote 0 packets"
        from "loop ran and wrote many packets" (which would mean the
        bug is downstream of pkt_write itself, e.g. seg_fd handling).*/
}AudSlot;

typedef struct{
    /* config */
    char    mcast[64],dir[256],name[64],iface[64];
    char    keyfile[256],keyuri[512],iv_hex[33];
    int     port; uint64_t start_seq; ChOpts opts;
    /* socket */
    int     fd;
    /* PSI — always write patched cached copy, never raw source */
    uint8_t pat[TS_SZ],pmt[TS_SZ];
    int     pat_ok,pmt_ok;
    uint16_t pmt_pid,vid_pid,pcr_pid;
    uint8_t  vid_stream_type;
    /* Multi-audio-track support: see AudSlot definition above Ch for
     * full rationale. n_aud_active is how many slots are in use.    */
    AudSlot  aud[MAX_AUD_TRACKS];
    int      n_aud_active;
    /* Audio PIDs discovered in the PMT but explicitly excluded by
     * audio_map (the unselected tracks that must be dropped from
     * output entirely, not merely omitted from PMT metadata).        */
    uint16_t aud_dropped[MAX_AUD_TRACKS];
    int      n_aud_dropped;
    /* SPS+PPS cache injected at every seg_open */
    uint8_t  spspps[8][TS_SZ];int spspps_n;
    /* segment */
    int     seg_fd; uint64_t seg_seq;
    double  dur[DUR_SLOTS]; uint64_t seg_size[DUR_SLOTS];
    int     target_dur,target_dur_fixed;
    /* AES */
    int     aes_on; Aes aes;
    /* write batch */
    uint8_t wb[WRITE_BATCH*TS_SZ]; int wb_n;
    /* timing */
    struct timespec seg_start; int seg_timing_ok;
    int64_t         last_pcr27;    /* last 27MHz PCR seen on pcr_pid,
        for interpolating PCR values into gap-stuffing packets.
        -1 = not yet seen (use plain null packets instead).           */
    /* Soft-bridge state for a genuine wall-clock STALE gap (age>=4s in
     * the stale-detection check below). See BRIDGE_GAP_MAX_SECS for
     * full rationale: real production gaps consistently measured
     * ~4.0-4.05s (a suspiciously exact, likely upstream-scheduled
     * outage, not random loss) were previously treated identically to
     * a genuine multi-minute channel failure -- full seg_close(),
     * c->started=0, PAT/PMT re-acquisition (CLEAN START), and a CC
     * discontinuity on resume -- which is what actually caused the
     * player-visible stop/restart, not the gap itself. A brief gap
     * below BRIDGE_GAP_MAX_SECS is now bridged in place (PCR-
     * interpolated stuffing, no segment/PAT/playlist disruption,
     * matching what the existing CC-drop path already does for much
     * smaller gaps) instead of escalating to the heavy path.          */
    int             bridging;          /* 1 while soft-bridging a gap */
    uint64_t        bridge_pkts_emitted; /* stuffing pkts written so far
        this bridge episode, for cumulative target-based pacing        */
    int             cc_drop_skip;  /* 1 = discard video PID packets until
        next PUSI after a CC gap, to avoid feeding a corrupted mid-PES
        continuation to the decoder. Set on CC drop in simple-copy mode,
        cleared on the next vid_pid packet with PUSI=1.               */
    /* Last time CC-drop recovery was actually triggered (decoder
     * flush + dec_ready reset) — used for a short cooldown so a burst
     * of many CC drops in the same fraction of a second doesn't keep
     * cancelling an in-progress recovery before the decoder gets a
     * real chance to produce output and let a segment open. See the
     * cooldown check at the CC-drop handler for full rationale.       */
    struct timespec last_recovery_trigger; int last_recovery_trigger_valid;
    /* state */
    int     started,recovering;
    /* CC */
    uint8_t cc_last[8192]; uint32_t cc_errors;
    /* PTS reset */
    int     pts_base_ok; int64_t pts_base;
    /* transcoders */
    AudXc   audxc; uint8_t aud_cc;
    AudScratch aud_scratch; /* shared scratch buffers for ALL of this
        channel's active audio tracks (see AudScratch doc for why this
        is safe to share: one channel's audxc_push calls across its
        tracks are always sequential, never concurrent). Lives here
        (once per channel) instead of inside AudXc (which would be
        once per track, up to MAX_AUD_TRACKS=8x more).                 */
    VidXc  *vidxc;  /* heap-allocated lazily, only for channels that
        set video_transcode=1 -- embedded by value (VidXc vidxc;) makes
        EVERY one of the MAX_CHANNELS=512 static g_ch[] slots cost
        ~1.3MB whether or not that channel uses video_transcode at all
        (due to VidXc's 1MB AuBuf, see AU_MAX): ~648MB committed at
        process startup unconditionally. As a pointer, channels that
        never set video_transcode=1 (the common case at high channel
        counts) cost only 8 bytes here instead of ~1MB; the real
        allocation only happens for channels that actually need it, in
        pkt_process's video_transcode init block. Every call site below
        is c->vidxc->field (NULL-guarded), not c->vidxc.field -- this
        has been converted back to a pointer at least once already
        after a copy-paste from an older snapshot reintroduced the
        by-value struct as a regression; if c->vidxc. (dot, not arrow)
        shows up anywhere again, that's the same regression back.    */
    /* stats */
    uint64_t dgrams,pkts,segs;
    struct timespec last_pkt;
    uint64_t snap_pkts,snap_bytes,snap_cc;
    struct timespec snap_time;
    EvtRing  evring;
    /* Recv/process decoupling ring — see nic_thread for full rationale.
     * Heap-allocated lazily on first use (same reasoning as vidxc
     * above: this struct is one of MAX_CHANNELS=512 static slots, so a
     * fixed-size array field here would cost RECV_RING_PKTS*TS_SZ bytes
     * per slot whether or not the channel is ever active).            */
    uint8_t *pkt_ring;         /* RECV_RING_PKTS * TS_SZ bytes, malloc'd
                                   lazily on first ring_push()          */
    int      ring_head,ring_tail,ring_count;
    uint64_t ring_drops;       /* packets dropped because the ring was
        full — this only happens if pkt_process (decode/encode) falls
        far enough behind recv() that RECV_RING_PKTS*TS_SZ worth of
        backlog piles up. Counted so it's visible in diagnostics
        instead of being silently indistinguishable from a genuine
        upstream/kernel-buffer drop (which shows up as a CC error).    */
}Ch;

static volatile sig_atomic_t g_stop=0;
static volatile sig_atomic_t g_dump_requested=0; /* set by SIGUSR1 handler;
    consumed by the stats thread, which dumps full per-channel audio/video
    state to help diagnose long-run issues (e.g. "audio silently stops
    after days, video stays fine, restart fixes it") that are impractical
    to reproduce on demand — trigger with `kill -USR1 <pid>` right before
    restarting, so the dump captures the actual failing state instead of
    a fresh post-restart one.                                            */
static Ch g_ch[MAX_CHANNELS]; static int g_nch=0;
static int g_del=KEEP_EXTRA;
static void onsig(int s){(void)s;g_stop=1;}
static void onsig_dump(int s){(void)s;g_dump_requested=1;}

/* Returns the index of the active audio slot whose out_pid (==src_pid,
 * per the design — see AudSlot) matches pid, or -1 if pid isn't one of
 * this channel's currently selected audio tracks. Used everywhere the
 * old code did a single `pid==c->aud_pid` check.                      */
static inline int aud_slot_for_pid(Ch*c,uint16_t pid){
    for(int i=0;i<c->n_aud_active;i++)
        if(c->aud[i].src_pid==pid)return i;
    return -1;}

/* True if pid is an audio track that was discovered in the PMT but
 * explicitly excluded by audio_map — its packets must be actively
 * dropped from output, not merely omitted from PMT metadata.         */
static inline int aud_is_dropped(Ch*c,uint16_t pid){
    for(int i=0;i<c->n_aud_dropped;i++)
        if(c->aud_dropped[i]==pid)return 1;
    return 0;}

/* ----------------------------------------------------------------
   TS HELPERS
   ---------------------------------------------------------------- */
static inline uint16_t ts_pid(const uint8_t*p){return((p[1]&0x1F)<<8)|p[2];}
static inline int ts_pusi(const uint8_t*p){return(p[1]>>6)&1;}
static inline int ts_afc(const uint8_t*p){return(p[3]>>4)&3;}
static inline int ts_poff(const uint8_t*p){
    int a=ts_afc(p),o;
    if(a==1)o=4;else if(a==3)o=5+(int)p[4];else return -1;
    return o<TS_SZ?o:-1;}

static uint16_t pat_parse(const uint8_t*p){
    int o=ts_poff(p);if(o<0||o>=TS_SZ)return 0;
    int b=o+1+p[o];if(b+8>=TS_SZ||p[b]!=0)return 0;
    int e=b+3+(((p[b+1]&0xF)<<8)|p[b+2])-4;
    for(int q=b+8;q+3<=e&&q+3<TS_SZ;q+=4){
        uint16_t pn=(p[q]<<8)|p[q+1];
        uint16_t pp=((p[q+2]&0x1F)<<8)|p[q+3];
        if(pn)return pp;}
    return 0;}
static int is_video(uint8_t t){return t==0x01||t==0x02||t==0x10||t==0x1B||t==0x24||t==0x27;}
static int is_audio(uint8_t t){return t==0x03||t==0x04||t==0x06||t==0x0F||t==0x11||t==0x81||t==0x87;}
/* Stream types our audio_transcode pipeline can actually decode:
 * 0x03/0x04 (MPEG-1/2 audio, MP2/MP3 via mpg123), 0x0F (already AAC, no
 * decode needed), and 0x06/0x87/0x81 (AC3/E-AC3, decoded via libavcodec
 * and downmixed to stereo if the source is 5.1/surround — see
 * audxc_push_ac3). NOTE on 0x81: this is a private-data tag that
 * different muxers use for different codecs (some use it for DTS, not
 * AC3) — treating it as AC3 here is confirmed correct for this
 * deployment's specific encoders, not a universally safe assumption.
 * HE-AAC LATM (0x11) remains unsupported: recognized as "audio" for
 * PMT parsing purposes but picking it when a decodable alternative
 * exists would silently feed unsupported bytes into a decoder that
 * can't handle them.                                                  */
static int is_decodable_audio(uint8_t t){return t==0x03||t==0x04||t==0x0F||t==0x06||t==0x87||t==0x81;}
static int pmt_parse(const uint8_t*p,uint16_t*pcr,uint16_t*vid,uint16_t*aud,uint8_t*vst,uint8_t*aud_st){
    int o=ts_poff(p);if(o<0||o+1>=TS_SZ)return 0;
    int b=o+1+p[o];if(b+12>=TS_SZ||p[b]!=0x02)return 0;
    *pcr=((p[b+8]&0x1F)<<8)|p[b+9];
    int pil=((p[b+10]&0xF)<<8)|p[b+11];
    int sec=((p[b+1]&0xF)<<8)|p[b+2];
    int es=b+12+pil,ee=b+3+sec-4;*vid=*aud=0;if(vst)*vst=0;if(aud_st)*aud_st=0;
    uint16_t fallback_aud=0;uint8_t fallback_st=0;
    while(es+4<TS_SZ&&es+4<=ee){
        uint8_t st=p[es];uint16_t sp=((p[es+1]&0x1F)<<8)|p[es+2];
        int el=((p[es+3]&0xF)<<8)|p[es+4];
        if(!*vid&&is_video(st)){*vid=sp;if(vst)*vst=st;}
        /* Prefer a decodable audio stream; remember the first audio
         * stream of ANY type as a fallback only in case nothing
         * decodable is found anywhere in the PMT.                    */
        if(!*aud&&is_decodable_audio(st)){*aud=sp;if(aud_st)*aud_st=st;}
        else if(!fallback_aud&&is_audio(st)){fallback_aud=sp;fallback_st=st;}
        es+=5+el;}
    if(!*aud&&fallback_aud){*aud=fallback_aud;if(aud_st)*aud_st=fallback_st;}
    return *pcr>0;}

typedef struct{
    uint16_t pid;
    uint8_t  stream_type;
}AudTrack;
/* Discover ALL audio elementary streams in the PMT, in the order they
 * appear (this order is what audio_map=N indexes into, 1-based, per
 * the position-based selection design). Returns the number found
 * (capped at MAX_AUD_TRACKS — real broadcasts essentially never have
 * more than a handful of audio tracks).                              */
static int pmt_parse_all_audio(const uint8_t*p,AudTrack*tracks){
    int o=ts_poff(p);if(o<0||o+1>=TS_SZ)return 0;
    int b=o+1+p[o];if(b+12>=TS_SZ||p[b]!=0x02)return 0;
    int pil=((p[b+10]&0xF)<<8)|p[b+11];
    int sec=((p[b+1]&0xF)<<8)|p[b+2];
    int es=b+12+pil,ee=b+3+sec-4;int n=0;
    while(es+4<TS_SZ&&es+4<=ee&&n<MAX_AUD_TRACKS){
        uint8_t st=p[es];uint16_t sp=((p[es+1]&0x1F)<<8)|p[es+2];
        int el=((p[es+3]&0xF)<<8)|p[es+4];
        if(is_audio(st)){tracks[n].pid=sp;tracks[n].stream_type=st;n++;}
        es+=5+el;}
    return n;}

static uint32_t crc32_mpeg(const uint8_t*d,int len){
    uint32_t crc=0xFFFFFFFF;
    for(int i=0;i<len;i++){
        crc^=(uint32_t)d[i]<<24;
        for(int b=0;b<8;b++)crc=(crc&0x80000000)?(crc<<1)^0x04C11DB7:(crc<<1);}
    return crc;}

/* Patch PMT audio stream_type to 0x0F (AAC), and strip any descriptors
 * attached to that elementary stream entry (e.g. a registration_
 * descriptor explicitly naming "AC-3" — common on real AC3 sources,
 * see rationale above this function). Leaving such a descriptor in
 * place after changing stream_type causes demuxers that trust
 * descriptors over the raw stream_type byte to keep misidentifying
 * the (now-AAC) stream by its original codec.
 * Handles 0x03 (MPEG-1 Audio), 0x04 (MPEG-2 Audio/MP2), and now
 * 0x06/0x87/0x81 (AC3/E-AC3, transcoded via the libavcodec-based AC3
 * decode path — see audxc_push_ac3).
 * Recalculates CRC32.                                          */
static int pmt_patch_audio(uint8_t*pkt,uint16_t aud_pid){
    int o=ts_poff(pkt);if(o<0||o+1>=TS_SZ)return 0;
    int b=o+1+pkt[o];if(b+12>=TS_SZ||pkt[b]!=0x02)return 0;
    int sec=((pkt[b+1]&0xF)<<8)|pkt[b+2];
    int pil=((pkt[b+10]&0xF)<<8)|pkt[b+11];
    int es=b+12+pil,ee=b+3+sec-4;int patched=0;int removed_bytes=0;
    while(es+4<TS_SZ&&es+4<=ee){
        uint8_t st=pkt[es];
        uint16_t sp=((pkt[es+1]&0x1F)<<8)|pkt[es+2];
        int el=((pkt[es+3]&0xF)<<8)|pkt[es+4];
        if(sp==aud_pid&&(st==0x03||st==0x04||st==0x06||st==0x87||st==0x81)){
            static int logged=0;
            if(!logged){
                printf("[pmt] audio pid=0x%04X type=0x%02X -> 0x0F (AAC)"
                       "%s\n",sp,st,el>0?" (descriptors stripped)":"");
                logged=1;}
            pkt[es]=0x0F;patched=1;
            if(el>0){
                /* Strip descriptors: shift everything after this
                 * entry's descriptor bytes left by el, closing the
                 * gap, and set this entry's ES_info_length to 0.      */
                int desc_start=es+5;
                int tail_start=desc_start+el;
                int tail_len=TS_SZ-tail_start;
                memmove(pkt+desc_start,pkt+tail_start,tail_len);
                memset(pkt+TS_SZ-el,0xFF,el);
                pkt[es+3]&=0xF0; /* ES_info_length high nibble -> 0 */
                pkt[es+4]=0x00;  /* ES_info_length low byte -> 0     */
                removed_bytes+=el;
                ee-=el;
                el=0; /* this entry's descriptors are gone now       */
            }}
        es+=5+el;}
    if(!patched)return 0;
    int new_sec=sec-removed_bytes;
    pkt[b+1]=(uint8_t)((pkt[b+1]&0xF0)|((new_sec>>8)&0xF));
    pkt[b+2]=(uint8_t)(new_sec&0xFF);
    int coff=b+3+new_sec-4;if(coff+4>TS_SZ)return 1;
    uint32_t crc=crc32_mpeg(pkt+b,new_sec-1);
    pkt[coff+0]=(crc>>24)&0xFF;pkt[coff+1]=(crc>>16)&0xFF;
    pkt[coff+2]=(crc>>8)&0xFF; pkt[coff+3]=(crc)&0xFF;
    return 1;}

/* Patch PMT video stream_type to 0x1B (H264) when video_transcode=1 */
/* Remove every audio elementary stream entry from the PMT except the
 * one whose PID equals keep_pid (use keep_pid=0 to remove ALL audio
 * entries, e.g. if audio_map pointed at a nonexistent track). Shifts
 * remaining bytes to close the gap and recomputes section_length+CRC.
 * Only ever removes bytes, so the result always fits in one 188-byte
 * TS packet — no multi-packet PMT handling is required.               */
static int pmt_strip_unselected_audio(uint8_t*pkt,uint16_t keep_pid){
    int o=ts_poff(pkt);if(o<0||o+1>=TS_SZ)return 0;
    int b=o+1+pkt[o];if(b+12>=TS_SZ||pkt[b]!=0x02)return 0;
    int sec=((pkt[b+1]&0xF)<<8)|pkt[b+2];
    int pil=((pkt[b+10]&0xF)<<8)|pkt[b+11];
    int es=b+12+pil,ee=b+3+sec-4;int removed_bytes=0;
    while(es+4<TS_SZ&&es+4<=ee){
        uint8_t st=pkt[es];
        uint16_t sp=((pkt[es+1]&0x1F)<<8)|pkt[es+2];
        int el=((pkt[es+3]&0xF)<<8)|pkt[es+4];
        int entry_len=5+el;
        if(is_audio(st)&&sp!=keep_pid){
            /* Remove this entry: shift everything after it left by
             * entry_len bytes, shrinking the section in place.        */
            int tail_start=es+entry_len;
            int tail_len=TS_SZ-tail_start; /* conservative — copy to end
                of packet; bytes past the real section end are stuffing
                /already-irrelevant and get naturally truncated by the
                shrunk section_length written below anyway.            */
            memmove(pkt+es,pkt+tail_start,tail_len);
            /* Zero the now-unused tail so no stale entry bytes linger
             * past the new (shorter) section in the packet buffer.    */
            memset(pkt+TS_SZ-entry_len,0xFF,entry_len);
            removed_bytes+=entry_len;
            ee-=entry_len;
            /* Don't advance es — the next entry has shifted into this
             * same position and must also be checked.                */
            continue;
        }
        es+=entry_len;}
    if(!removed_bytes)return 0;
    int new_sec=sec-removed_bytes;
    pkt[b+1]=(uint8_t)(((pkt[b+1]&0xF0))|((new_sec>>8)&0xF));
    pkt[b+2]=(uint8_t)(new_sec&0xFF);
    int coff=b+3+new_sec-4;if(coff+4>TS_SZ)return 1;
    uint32_t crc=crc32_mpeg(pkt+b,new_sec-1);
    pkt[coff+0]=(crc>>24)&0xFF;pkt[coff+1]=(crc>>16)&0xFF;
    pkt[coff+2]=(crc>>8)&0xFF; pkt[coff+3]=(crc)&0xFF;
    return 1;}


/* Patch PMT video stream_type to 0x1B (H264) when video_transcode=1 */
static int pmt_patch_video(uint8_t*pkt,uint16_t vid_pid){
    int o=ts_poff(pkt);if(o<0||o+1>=TS_SZ)return 0;
    int b=o+1+pkt[o];if(b+12>=TS_SZ||pkt[b]!=0x02)return 0;
    int sec=((pkt[b+1]&0xF)<<8)|pkt[b+2];
    int pil=((pkt[b+10]&0xF)<<8)|pkt[b+11];
    int es=b+12+pil,ee=b+3+sec-4;int patched=0;
    while(es+4<TS_SZ&&es+4<=ee){
        uint8_t st=pkt[es];
        uint16_t sp=((pkt[es+1]&0x1F)<<8)|pkt[es+2];
        int el=((pkt[es+3]&0xF)<<8)|pkt[es+4];
        if(sp==vid_pid&&st!=0x1B){pkt[es]=0x1B;patched=1;}
        es+=5+el;}
    if(!patched)return 0;
    int coff=b+3+sec-4;if(coff+4>TS_SZ)return 1;
    uint32_t crc=crc32_mpeg(pkt+b,sec-1);
    pkt[coff+0]=(crc>>24)&0xFF;pkt[coff+1]=(crc>>16)&0xFF;
    pkt[coff+2]=(crc>>8)&0xFF; pkt[coff+3]=(crc)&0xFF;
    return 1;}

/* H264 NAL scanner. Scans ALL NALs in the packet payload.
 * Returns IDR(5/20) immediately if found — highest priority.
 * Skips AUD(9) and SEI(6) entirely — they precede every frame
 * and must not mask the real IDR/SPS/PPS that follows them.
 * Returns SPS(7), PPS(8), or other NAL type as fallback.
 * Returns -1 if no H264 NAL found.                           */
static int nal_type(const uint8_t*pkt){
    int off=ts_poff(pkt);if(off<0)return -1;
    const uint8_t*pay=pkt+off;int plen=TS_SZ-off;
    if(ts_pusi(pkt)){
        if(plen<9||pay[0]!=0||pay[1]!=0||pay[2]!=1)return -1;
        int skip=9+pay[8];if(skip>=plen)return -1;
        pay+=skip;plen-=skip;}
    int best=-1;
    for(int i=0;i+3<plen;i++){
        int nt=-1;
        if(pay[i]==0&&pay[i+1]==0&&pay[i+2]==0&&i+4<plen&&pay[i+3]==1)
            {nt=pay[i+4]&0x1F;i+=4;}
        else if(pay[i]==0&&pay[i+1]==0&&pay[i+2]==1&&i+3<plen)
            {nt=pay[i+3]&0x1F;i+=3;}
        if(nt<0)continue;
        if(nt==5||nt==20)return nt;   /* IDR: return immediately */
        if(nt==9||nt==6)continue;     /* AUD/SEI: skip, keep scanning */
        if(nt==7&&best!=5&&best!=20)best=nt;  /* SPS */
        else if(nt==8&&best<0)best=nt;         /* PPS */
        else if(best<0)best=nt;
    }
    return best;}
static int has_idr(const uint8_t*pkt){int n=nal_type(pkt);return n==5||n==20;}

/* Scan a TS packet's payload for the first slice NAL (type 1 or 5) and
 * apply the real Exp-Golomb slice_type check (is_islice) rather than
 * just trusting nal_type()'s raw NAL-type classification. Some real
 * broadcast encoders (confirmed on this exact source family) use
 * ordinary non-IDR NAL units (type 1) for what are functionally
 * keyframes — relying on nal_type()==5 alone means segment cuts almost
 * never fire on such sources, since they essentially never emit a
 * real type-5 IDR. This is the same detection already used in the
 * video_transcode re-encode path (vidxc_push) — applying it here too
 * fixes segment-cut reliability for the simple-copy/audio_transcode-
 * only path on these sources, without needing video_transcode=1.      */
static int has_keyframe(const uint8_t*pkt){
    int off=ts_poff(pkt);if(off<0)return 0;
    const uint8_t*pay=pkt+off;int plen=TS_SZ-off;
    if(ts_pusi(pkt)){
        if(plen<9||pay[0]!=0||pay[1]!=0||pay[2]!=1)return 0;
        int skip=9+pay[8];if(skip>=plen)return 0;
        pay+=skip;plen-=skip;}
    for(int i=0;i+3<plen;i++){
        int nt=-1;int hdr_off=-1;
        if(pay[i]==0&&pay[i+1]==0&&pay[i+2]==0&&i+4<plen&&pay[i+3]==1)
            {nt=pay[i+4]&0x1F;hdr_off=i+4;}
        else if(pay[i]==0&&pay[i+1]==0&&pay[i+2]==1)
            {nt=pay[i+3]&0x1F;hdr_off=i+3;}
        else continue;
        if(nt==1||nt==5){
            int payload_off=hdr_off+1;
            return is_islice(pay+payload_off,plen-payload_off);
        }
    }
    return 0;}

/* Rewrite the first slice NAL (type 1) found in this packet's payload
 * to type 5 (IDR), in place. ONLY called on the exact packet already
 * identified as a cut point via has_keyframe() (i.e. is_islice()
 * already confirmed this slice is genuinely all-intra) — this is a
 * targeted fix for sources that never emit real type-5 IDR NALs (see
 * has_keyframe/is_islice docs for the "functional keyframe" quirk;
 * confirmed via direct inspection of real captured segments: zero NAL
 * type 5 anywhere across 8 consecutive real production segments, only
 * type-1 slices satisfying the Exp-Golomb I-slice check).
 *
 * Rationale: nal_unit_type alone is what tells a decoder "discard all
 * prior reference pictures, this is a clean random-access point" —
 * an Exp-Golomb-confirmed all-intra type-1 slice has the right PICTURE
 * CONTENT for that (no macroblock in it references another frame) but
 * the type-1 tag itself gives the decoder no such instruction, so its
 * reference-picture-buffer bookkeeping (mmco/short-term/long-term ref
 * tracking) carries over from before the segment boundary — exactly
 * what produced the "mmco: unref short failure"/"reference picture
 * missing" errors confirmed on EVERY one of 8 real segments inspected,
 * since any standalone-decoded segment (which is what HLS player-side
 * per-segment decoding effectively is) has no real prior frames for
 * that leftover bookkeeping to refer to. Setting nal_unit_type=5 here
 * is a single bit flip in the NAL header byte (0x01->0x05 in the low
 * 5 bits) and does not touch the slice payload bits at all — the
 * already-intra-coded picture content is unaffected either way.
 *
 * Known residual risk (acknowledged, not fully resolved by this fix):
 * a real decoder seeing nal_unit_type=5 typically also resets its
 * internal frame_num/POC tracking to IDR-relative expectations, while
 * the bitstream's own frame_num field (inside the slice header, not
 * touched by this patch) continues from the prior segment's numbering
 * rather than restarting at the spec-implied baseline for a true IDR.
 * This is a real, unverified-against-a-live-decoder gap — if a
 * regression appears (e.g. a NEW class of artifact right at segment
 * boundaries that wasn't there before), this function is the first
 * thing to disable via idr_rewrite=0.                                 */
static int rewrite_slice_to_idr(uint8_t*pkt){
    int off=ts_poff(pkt);if(off<0)return 0;
    uint8_t*pay=pkt+off;int plen=TS_SZ-off;
    if(ts_pusi(pkt)){
        if(plen<9||pay[0]!=0||pay[1]!=0||pay[2]!=1)return 0;
        int skip=9+pay[8];if(skip>=plen)return 0;
        pay+=skip;plen-=skip;}
    for(int i=0;i+3<plen;i++){
        int nt=-1;int hdr_off=-1;
        if(pay[i]==0&&pay[i+1]==0&&pay[i+2]==0&&i+4<plen&&pay[i+3]==1)
            {nt=pay[i+4]&0x1F;hdr_off=i+4;}
        else if(pay[i]==0&&pay[i+1]==0&&pay[i+2]==1)
            {nt=pay[i+3]&0x1F;hdr_off=i+3;}
        else continue;
        if(nt==1){
            /* Rewrite low 5 bits of the NAL header byte: type 1 -> 5.
             * nal_ref_idc (bits 6-5) and forbidden_zero_bit (bit 7)
             * are left exactly as the source set them.                */
            int ts_byte=(int)(pay-pkt-off)+hdr_off+off;
            if(ts_byte<TS_SZ)
                pkt[ts_byte]=(uint8_t)((pkt[ts_byte]&0xE0)|0x05);
            return 1;
        }
        if(nt==5)return 0; /* already a real IDR, nothing to do */
    }
    return 0;}

/* SPS progressive patch: set frame_mbs_only_flag=1 for LG/Samsung */
static int sps_patch_interlace(uint8_t*pkt){
    int off=ts_poff(pkt);if(off<0)return 0;
    uint8_t*pay=pkt+off;int plen=TS_SZ-off;
    if(ts_pusi(pkt)){
        if(plen<9||pay[0]!=0||pay[1]!=0||pay[2]!=1)return 0;
        int skip=9+pay[8];if(skip>=plen)return 0;
        pay+=skip;plen-=skip;}
    for(int i=0;i+3<plen;i++){
        int n=-1;
        if(pay[i]==0&&pay[i+1]==0&&pay[i+2]==1)n=i+3;
        else if(i+4<plen&&pay[i]==0&&pay[i+1]==0&&pay[i+2]==0&&pay[i+3]==1)n=i+4;
        if(n<0)continue;if(n>=plen)break;
        if((pay[n]&0x1F)!=7){i=n;continue;}
        const uint8_t*sps=pay+n+1;int slen=plen-n-1;
        if(slen<6)break;
        uint8_t profile=sps[0]; int bit=24;
        #define RB(nb)({uint32_t _v=0;for(int _b=0;_b<(nb);_b++){int _y=(bit+_b)/8,_i=7-((bit+_b)%8);if(_y<slen)_v=(_v<<1)|((sps[_y]>>_i)&1);else _v<<=1;}bit+=(nb);_v;})
        #define EG()({int _z=0;while(_z<32){int _y=(bit+_z)/8,_i=7-((bit+_z)%8);if(_y<slen&&!((sps[_y]>>_i)&1))_z++;else break;}bit+=_z+1;uint32_t _v=(1u<<_z)-1;for(int _b=0;_b<_z;_b++){int _y=(bit+_b)/8,_i=7-((bit+_b)%8);if(_y<slen)_v=(_v<<1)|((sps[_y]>>_i)&1);else _v<<=1;}bit+=_z;_v;})
        if(profile==100||profile==110||profile==122||profile==244||
           profile==44||profile==83||profile==86||profile==118||profile==128){
            uint32_t chroma=EG();if(chroma==3)RB(1);
            EG();EG();RB(1);
            if(RB(1)){int nl=(chroma!=3)?8:12;
                for(int sl=0;sl<nl;sl++)
                    if(RB(1)){int sz=(sl<6)?16:64;int last=8,next=8;
                        for(int j=0;j<sz;j++)if(next){int d=(int)EG();next=(last+(d%256)+256)%256;last=next;}}}}
        EG();EG();
        {uint32_t poc=EG();
         if(poc==0){EG();}
         else if(poc==1){RB(1);EG();EG();uint32_t n2=EG();for(uint32_t j=0;j<n2;j++)EG();}}
        EG();RB(1);EG();EG();
        int fb=(bit)/8,fbi=7-(bit%8);
        #undef RB
        #undef EG
        if(fb>=slen)break;
        int ts_byte=(int)(pay-pkt-off)+n+1+fb+off;
        if(ts_byte>=TS_SZ)break;
        if((pkt[ts_byte]>>fbi)&1)break;
        pkt[ts_byte]|=(1<<fbi);
        return 1;}
    return 0;}

static int64_t pes_read_pts(const uint8_t*p){
    return(int64_t)(((uint64_t)(p[0]&0x0E)<<29)|((uint64_t)p[1]<<22)|
                    ((uint64_t)(p[2]&0xFE)<<14)|((uint64_t)p[3]<<7)|
                    ((uint64_t)(p[4]&0xFE)>>1));}
static void pes_write_pts(uint8_t*p,int64_t v,uint8_t m4){
    p[0]=(uint8_t)(m4|((v>>29)&0x0E));p[1]=(uint8_t)((v>>22)&0xFF);
    p[2]=(uint8_t)(0x01|((v>>14)&0xFE));p[3]=(uint8_t)((v>>7)&0xFF);
    p[4]=(uint8_t)(0x01|((v<<1)&0xFE));}
static void pes_patch_pts(uint8_t*pkt,int64_t*base,int*ok){
    if(!ts_pusi(pkt))return;
    int o=ts_poff(pkt);if(o<0||o+8>=TS_SZ)return;
    const uint8_t*pes=pkt+o;
    if(pes[0]!=0||pes[1]!=0||pes[2]!=1)return;
    uint8_t pf=(pes[7]>>6)&3;if(!pf)return;
    int po=o+9;if(po+5>TS_SZ)return;
    int64_t pts=pes_read_pts(pkt+po);
    if(!*ok){*base=pts;*ok=1;}
    int64_t np=pts-*base;if(np<0)np=0;
    pes_write_pts(pkt+po,np,(pf==3)?0x31:0x21);
    if(pf==3){int dp=po+5;if(dp+5<=TS_SZ){
        int64_t dts=pes_read_pts(pkt+dp);
        int64_t nd=dts-*base;if(nd<0)nd=0;
        pes_write_pts(pkt+dp,nd,0x11);}}}

/* ----------------------------------------------------------------
   WRITE / SEGMENT / PLAYLIST
   ---------------------------------------------------------------- */
static void wb_flush(Ch*c){
    if(!c->wb_n||c->seg_fd<0)return;
    ssize_t r=write(c->seg_fd,c->wb,(size_t)c->wb_n*TS_SZ);(void)r;
    c->wb_n=0;}
static void pkt_write(Ch*c,const uint8_t*p){
    if(c->seg_fd<0)return;
    if(c->aes_on)aes_push(&c->aes,c->seg_fd,p);
    else{memcpy(c->wb+c->wb_n*TS_SZ,p,TS_SZ);if(++c->wb_n>=WRITE_BATCH)wb_flush(c);}}

static void write_m3u8(Ch*c){
    uint64_t complete=c->seg_seq-c->start_seq;
    if(complete<3)return;
    if(!c->target_dur_fixed){c->target_dur=SEGMENT_SECS+2;c->target_dur_fixed=1;}
    uint64_t win=complete>(uint64_t)MAX_SEGMENTS?(uint64_t)MAX_SEGMENTS:complete;
    uint64_t seq0=c->seg_seq-win;
    char tmp[512],fin[512];
    snprintf(fin,512,"%s/%s.m3u8",c->dir,c->name);
    snprintf(tmp,512,"%s/%s.m3u8.tmp",c->dir,c->name);
    FILE*f=fopen(tmp,"w");if(!f)return;
    fprintf(f,"#EXTM3U\n#EXT-X-VERSION:3\n"
              "#EXT-X-TARGETDURATION:%d\n"
              "#EXT-X-MEDIA-SEQUENCE:%llu\n",
              c->target_dur,(unsigned long long)seq0);
    if(c->keyuri[0])
        fprintf(f,"#EXT-X-KEY:METHOD=AES-128,URI=\"%s\",IV=0x%s\n",
                c->keyuri,c->iv_hex);
    for(uint64_t s=seq0;s<c->seg_seq;s++){
        double d=c->dur[s%DUR_SLOTS];
        if(d<=0)d=(double)SEGMENT_SECS;
        if(d>(double)c->target_dur)d=(double)c->target_dur;
        fprintf(f,"#EXTINF:%.3f,\nindex%llu.ts\n",d,(unsigned long long)s);}
    fflush(f);fclose(f);rename(tmp,fin);}

static void seg_close(Ch*c,const struct timespec*now){
    if(c->seg_fd<0)return;
    if(c->aes_on)aes_flush(&c->aes,c->seg_fd,1);else wb_flush(c);
    uint64_t fsize=0;{struct stat st;if(fstat(c->seg_fd,&st)==0)fsize=st.st_size;}
    close(c->seg_fd);c->seg_fd=-1;
    double el=c->seg_timing_ok?wall_el(&c->seg_start,now):0.0;
    uint64_t cl=c->seg_seq-1;
    c->dur[cl%DUR_SLOTS]=el;c->seg_size[cl%DUR_SLOTS]=fsize;c->segs++;
    {uint32_t em=(uint32_t)(el*1000);uint32_t kbps=el>0?(uint32_t)(fsize*8/el/1000):0;
     Evt e={EVT_CLOSE,time(NULL),cl,em,kbps,0,0,0};evpush(&c->evring,&e);}
    write_m3u8(c);}

/* Abort a segment damaged by a CC drop without publishing it.
 *
 * This is the simple-copy-mode alternative to seg_close() on CC error.
 * The difference: seg_close() calls write_m3u8(), which immediately
 * publishes the segment to the player's playlist — so a short, damaged
 * segment (e.g. 0.059s, confirmed from real logs) becomes visible to
 * ExoPlayer, which then tries to decode a truncated or mid-GOP
 * bitstream and freezes or shows macroblocks until the next clean IDR
 * arrives. seg_abort() instead:
 *
 *   1. Flushes and closes the file (so OS resources are released cleanly)
 *   2. Deletes the file from disk (no partial segment left behind)
 *   3. Rolls back seg_seq by 1 (so the next seg_open() reuses this slot)
 *   4. Does NOT call write_m3u8() (player never sees this segment)
 *   5. Does NOT increment c->segs (doesn't count as a produced segment)
 *
 * The net effect from the player's perspective: the previous good
 * segment is the last thing it sees, followed eventually by a new
 * clean segment starting on an IDR. No broken intermediate segment
 * appears in the playlist at all. The player may briefly stall waiting
 * for the next segment (the recovery time until an IDR arrives — at
 * 5.5Mbps with a 2s GOP, typically 0-2s), but a brief stall is
 * dramatically better than a freeze-on-broken-segment or macroblock
 * burst that ExoPlayer can take several seconds to recover from.
 *
 * Only used for simple-copy mode (video_transcode=0). The transcode
 * path already has its own CC recovery via the encoder's own IDR
 * re-emission cycle, which handles this differently.                  */
static void seg_abort(Ch*c){
    if(c->seg_fd<0)return;
    /* Flush+close without calling write_m3u8 */
    if(c->aes_on)aes_flush(&c->aes,c->seg_fd,1);else wb_flush(c);
    close(c->seg_fd);c->seg_fd=-1;
    /* Delete the partial file — don't leave a broken segment on disk */
    char path[512];
    snprintf(path,512,"%s/index%llu.ts",c->dir,
             (unsigned long long)(c->seg_seq-1));
    unlink(path);
    /* Roll back seg_seq so the next seg_open() reuses this slot number.
     * This keeps the playlist sequence gap-free: the aborted segment
     * slot is simply never published, and the next clean segment takes
     * its place as if the abort never happened.                        */
    c->seg_seq--;
    c->seg_timing_ok=0;
    /* Log the abort so it's visible in the event ring / stats output,
     * distinct from a normal seg_close. Uses EVT_CLOSE with duration=0
     * as a proxy since there's no dedicated EVT_ABORT type — the 0ms
     * duration distinguishes it from a real close in post-analysis.   */
    {Evt e={EVT_CLOSE,time(NULL),c->seg_seq,0,0,0,0,0};
     evpush(&c->evring,&e);}}

static void seg_open_ex(Ch*c,const struct timespec*now,int suppress_replay){
    int keep=MAX_SEGMENTS+g_del;
    if(c->seg_seq>=(uint64_t)keep+c->start_seq){
        char dp[512];
        snprintf(dp,512,"%s/index%llu.ts",c->dir,
                 (unsigned long long)(c->seg_seq-(uint64_t)keep));
        unlink(dp);}
    char path[512];
    snprintf(path,512,"%s/index%llu.ts",c->dir,(unsigned long long)c->seg_seq);
    c->seg_fd=open(path,O_WRONLY|O_CREAT|O_TRUNC,0644);
    if(c->seg_fd<0)return;
    if(c->aes_on)aes_reset(&c->aes);
    c->wb_n=0;c->seg_seq++;c->seg_start=*now;c->seg_timing_ok=1;
    /* Write patched PAT+PMT then cached SPS+PPS — unless the caller
     * already knows the very next packet it writes supplies fresh
     * SPS+PPS itself (suppress_replay), in which case replaying the
     * cache here would produce a duplicate parameter-set sequence at
     * the segment boundary (see call site in the segment-cut block
     * for the full rationale and the real corruption this caused).   */
    if(c->pat_ok)pkt_write(c,c->pat);
    if(c->pmt_ok)pkt_write(c,c->pmt);
    if(!suppress_replay)
        for(int i=0;i<c->spspps_n;i++)pkt_write(c,c->spspps[i]);
    {Evt e={EVT_OPEN,time(NULL),c->seg_seq-1,0,0,0,0,0};evpush(&c->evring,&e);}}
static void seg_open(Ch*c,const struct timespec*now){
    seg_open_ex(c,now,0);}

/* ----------------------------------------------------------------
   PACKET PROCESSOR
   ---------------------------------------------------------------- */
static void pkt_process(Ch*c,const uint8_t*p,const struct timespec*now){
    if(p[0]!=0x47)return;
    uint16_t pid=ts_pid(p);
    if(pid==0x1FFF)return;
    if(c->n_aud_dropped&&aud_is_dropped(c,pid))return; /* audio_map: this
        track was explicitly excluded — drop its packets entirely, not
        just its PMT entry (see aud_is_dropped doc for full rationale) */
    c->pkts++;c->snap_bytes+=TS_SZ;
    /* Track last PCR for gap-stuffing interpolation. Updated here,
     * before CC-drop handling, so we always have the most recent
     * 27MHz PCR available when a gap is detected just below.        */
    if(c->pcr_pid&&pid==c->pcr_pid){
        int64_t pcr=ts_read_pcr(p);
        if(pcr>=0)c->last_pcr27=pcr;}

    /* -- PAT -------------------------------------------------- */
    if(pid==0x0000){
        if(!c->pat_ok){
            uint16_t pp=pat_parse(p);
            if(pp){c->pmt_pid=pp;c->pat_ok=1;
                Evt e={EVT_PAT,time(NULL),0,0,0,pp,0,0};evpush(&c->evring,&e);}}
        if(c->pat_ok)memcpy(c->pat,p,TS_SZ);
        /* Write patched PAT (not raw source) */
        if(c->pat_ok&&c->started)pkt_write(c,c->pat);
        return;}

    /* -- PMT -------------------------------------------------- */
    if(c->pat_ok&&pid==c->pmt_pid){
        uint16_t vp=0,ap_unused=0,cp=0;uint8_t vst=0,aud_st_unused=0;
        AudTrack disc[MAX_AUD_TRACKS];
        int n_disc=pmt_parse_all_audio(p,disc);
        /* pmt_parse still used for video PID + PCR PID — those remain
         * single-value by nature (a program has one PCR PID and, for
         * our purposes, one video PID), so no need to touch that path. */
        if(pmt_parse(p,&cp,&vp,&ap_unused,&vst,&aud_st_unused)&&cp){
            if(!c->pmt_ok||cp!=c->pcr_pid||vp!=c->vid_pid){
                c->pcr_pid=cp;c->vid_pid=vp;c->vid_stream_type=vst;c->pmt_ok=1;
                /* Determine which discovered audio track(s) to select. */
                int sel_lo=0,sel_hi=n_disc; /* default: select ALL tracks */
                if(c->opts.audio_map>0){
                    if(c->opts.audio_map<=n_disc){sel_lo=c->opts.audio_map-1;sel_hi=c->opts.audio_map;}
                    else{sel_lo=sel_hi=0; /* requested index doesn't exist */
                        fprintf(stderr,"[%s] WARNING: audio_map=%d requested but "
                                "only %d audio track(s) found in PMT — no audio "
                                "track selected.\n",c->name,c->opts.audio_map,n_disc);}
                }
                c->n_aud_active=0;c->n_aud_dropped=0;
                for(int i=0;i<n_disc;i++){
                    int selected=(i>=sel_lo&&i<sel_hi);
                    if(!selected){
                        if(c->opts.audio_map>0&&c->n_aud_dropped<MAX_AUD_TRACKS)
                            c->aud_dropped[c->n_aud_dropped++]=disc[i].pid;
                        continue;
                    }
                    if(c->n_aud_active>=MAX_AUD_TRACKS)continue;
                    AudSlot*s=&c->aud[c->n_aud_active];
                    s->src_pid=s->out_pid=disc[i].pid;
                    s->stream_type=disc[i].stream_type;
                    c->n_aud_active++;
                    if(c->opts.audio_transcode){
                        if(!is_decodable_audio(disc[i].stream_type))
                            fprintf(stderr,"[%s] WARNING: audio_transcode requested but "
                                    "audio track %d (PID 0x%04X) has stream_type=0x%02X "
                                    "(not MP2/MP3/AAC) — our decoder cannot handle this; "
                                    "this track will be passed through untouched.\n",
                                    c->name,i+1,disc[i].pid,disc[i].stream_type);
                        else if(disc[i].stream_type==0x0F)
                            printf("[%s] audio track %d (PID 0x%04X) is already AAC — "
                                   "audio_transcode skipped for it, passing through as-is\n",
                                   c->name,i+1,disc[i].pid);
                    }
                }
                Evt e={EVT_PMT,time(NULL),0,(uint32_t)(c->n_aud_active?c->aud[0].src_pid:0),
                       (uint32_t)cp,vp,0,0};
                evpush(&c->evring,&e);}}
        /* Cache and patch PMT — NEVER write raw source PMT to segments.
         * Patch stream_type for every track we'll transcode to AAC, and
         * (for audio_map mode) strip any audio track NOT selected, per
         * the design: unselected tracks are dropped entirely from the
         * output, not merely left untranscoded.                        */
        memcpy(c->pmt,p,TS_SZ);
        if(c->opts.audio_map>0)
            pmt_strip_unselected_audio(c->pmt,c->n_aud_active?c->aud[0].src_pid:0);
        if(c->opts.fix_mp2||c->opts.audio_transcode)
            for(int i=0;i<c->n_aud_active;i++)
                pmt_patch_audio(c->pmt,c->aud[i].src_pid);
        if(c->opts.video_transcode&&c->vid_pid)
            pmt_patch_video(c->pmt,c->vid_pid);
        /* Init audio transcoder for each selected track that needs one
         * (genuinely decodable source — MP2/MP3 via mpg123, or AC3/
         * E-AC3 via libavcodec; already-AAC tracks are passed through
         * with no transcoder, per the audxc.active guard in the
         * per-packet dispatch below).                                  */
        if(c->opts.audio_transcode)
            for(int i=0;i<c->n_aud_active;i++){
                AudSlot*s=&c->aud[i];
                if(s->stream_type==0x0F||!is_decodable_audio(s->stream_type))continue;
                if(s->audxc.active)continue;
                AudXcSrc src_codec=(s->stream_type==0x06||s->stream_type==0x87||
                                     s->stream_type==0x81)?
                                    AUDXC_SRC_AC3:AUDXC_SRC_MP2;
                printf("[%s] audio_transcode: track%d src_pid=0x%04X codec=%s "
                       "bitrate=%d coder=%s\n",
                       c->name,i+1,s->src_pid,
                       src_codec==AUDXC_SRC_AC3?"AC3->AAC":"MP2->AAC",
                       c->opts.audio_bitrate>0?c->opts.audio_bitrate:128000,
                       c->opts.audio_coder);
                if(!audxc_init(&s->audxc,c->opts.audio_bitrate,c->opts.audio_coder,src_codec))
                    fprintf(stderr,"[%s] audxc_init FAILED for track%d\n",c->name,i+1);
                s->cc=0;s->audxc.pusi_seen=0;}
        /* Init video transcoder once -- VidXc is heap-allocated here,
         * lazily, only for channels that actually set video_transcode=1
         * (see the VidXc* field doc in Ch for why).                    */
        if(c->opts.video_transcode&&c->vid_pid&&!c->vidxc){
            c->vidxc=calloc(1,sizeof(VidXc));
            if(!c->vidxc){
                fprintf(stderr,"[%s] video_transcode: out of memory\n",c->name);
            } else {
                uint8_t st=c->vid_stream_type?c->vid_stream_type:0x1B;
                float crf=c->opts.video_crf>0?c->opts.video_crf:23.0f;
                if(vidxc_init(c->vidxc,st,25,1,crf))
                    printf("[%s] video_transcode: 0x%02X crf=%.0f\n",c->name,st,crf);
                else {
                    /* init failed -- free immediately so we retry
                     * cleanly next time this block is reached rather
                     * than leaking a half-initialized struct forever. */
                    free(c->vidxc); c->vidxc=NULL;
                }
            }
        }
        /* Write patched PMT to segment */
        if(c->started)pkt_write(c,c->pmt);
        return;}

    /* -- CC drop detection ------------------------------------ */
    /* Only video/audio PIDs trigger seg_close+recovery — a CC blip on
     * some other PID (SDT, EIT, or any other metadata/PSI PID we don't
     * even forward into the output) is completely harmless to playback
     * since we never write those packets to a segment anyway. Treating
     * every PID's CC discontinuity as disruptive caused frequent,
     * spurious segment truncation/resync on real broadcast sources —
     * confirmed via a 70s continuous test: a periodic CC blip on PID
     * 0x0011 (SDT) was forcing seg_close()+recovering=1 roughly every
     * ~10s, producing truncated segments (as short as 0.348s) and a
     * brief resync gap each time — closely matching reports of the
     * stream "stopping every 10 seconds" in VLC and on an LG TV. We
     * still count/track CC errors on every PID for stats visibility,
     * but only act on them for PIDs that actually end up in the output.*/
    {int afc=ts_afc(p);
     if((afc==1||afc==3)&&pid!=0x1FFF&&pid!=0x0000&&pid!=c->pmt_pid){
         uint8_t cc=(uint8_t)(p[3]&0x0F);
         uint8_t prev=c->cc_last[pid&0x1FFF];
         if(prev!=0xFF){
             uint8_t exp=(uint8_t)((prev+1)&0x0F);
             if(cc!=exp&&cc!=prev){
                 c->cc_errors++;
                 int relevant=(pid==c->vid_pid)||(aud_slot_for_pid(c,pid)>=0);
                 if(relevant&&c->started){
                     Evt e={EVT_CC,time(NULL),0,0,0,(uint16_t)(pid&0x1FFF),exp,cc};
                     evpush(&c->evring,&e);
                     /* For a VIDEO-PID drop under video_transcode, decide
                      * the cooldown ONCE and gate seg_close together with
                      * the decoder-reset actions below — otherwise the
                      * segment still gets torn down on every drop in a
                      * burst even though the decoder reset is correctly
                      * throttled, and reopening is gated on the (now
                      * recovering) decoder producing a keyframe, so the
                      * gap persists for the same reason just one step
                      * removed. Confirmed via direct re-analysis of real
                      * production logs taken AFTER the first cooldown
                      * attempt: the same multi-second gaps continued,
                      * because this exact call was still unconditional.
                      * Audio-only drops (or video drops when
                      * video_transcode isn't active) keep the original
                      * always-close behavior — the cooldown is
                      * specifically about not interrupting an in-progress
                      * VIDEO recovery, since video decode/keyframe
                      * detection is the real bottleneck, not segment
                      * file I/O.                                        */
                     int video_recovery_throttled=0;
                     if(c->opts.video_transcode&&pid==c->vid_pid){
                         double since_last=c->last_recovery_trigger_valid?
                             wall_el(&c->last_recovery_trigger,now):1e9;
                         video_recovery_throttled=(since_last<0.7);
                     }
                     if(c->seg_fd>=0&&!video_recovery_throttled){
                         /* Simple-copy mode: abort the damaged segment
                          * rather than closing it normally. seg_abort()
                          * deletes the partial file and rolls back
                          * seg_seq without calling write_m3u8() — so
                          * ExoPlayer never sees this broken segment in
                          * the playlist at all. The player gets a brief
                          * stall (time until the next clean IDR) instead
                          * of the current behaviour (a very short,
                          * mid-GOP segment ? freeze/macroblock that can
                          * persist for several seconds). Confirmed root
                          * cause of the drops: disk I/O stalls on /dev/
                          * sda at 98.4% utilisation causing kernel UDP
                          * receive-buffer overflows (158M errors seen in
                          * netstat -su), dropping entire 1316-byte UDP
                          * datagrams (7 TS packets at once — exactly
                          * matching the observed CC gap sizes of 6-7).
                          * For video_transcode mode, the original
                          * seg_close() is kept since that path manages
                          * its own recovery via the encoder's own IDR
                          * re-emission cycle rather than this state
                          * machine.                                     */
                         /* Simple-copy video PID drop — null-packet
                          * stuffing instead of segment abort/restart.
                          *
                          * Root cause context: disk I/O stalls on the
                          * segment write path (sda at 98.4% utilisation,
                          * 472ms avg await) cause the kernel UDP receive
                          * buffer to fill and drop entire 1316-byte UDP
                          * datagrams (7 TS packets each — confirmed:
                          * 158M receive buffer errors in netstat -su,
                          * CC gaps consistently 6-7 packets per event).
                          *
                          * What TSDuck does (confirmed by comparing the
                          * same source piped through TSDuck?ffmpeg vs
                          * udp_hls direct): on detecting a CC gap, it
                          * writes (gap) null TS packets (PID=0x1FFF) to
                          * fill the missing slots, maintaining stream
                          * timing/continuity. The H.264 decoder conceals
                          * the missing frame(s) — at most a brief
                          * macroblock for ~2ms — then resumes normally.
                          * NO segment restart, NO IDR wait, NO stall.
                          *
                          * What the old approach did (seg_abort then
                          * recovering=1): deleted the current segment
                          * from the playlist, then blocked all video
                          * writes until the next IDR (0-2s). ExoPlayer
                          * saw a gap in the playlist ? stall/freeze for
                          * the full IDR wait period, far worse than the
                          * decoder-level concealment TSDuck achieves.
                          *
                          * The null stuffing cap (16 max = one full CC
                          * cycle) prevents runaway stuffing if the gap
                          * value is somehow corrupted or wraps; normal
                          * disk-stall drops produce gaps of 6-7.       */
                         if(!c->opts.video_transcode&&pid==c->vid_pid
                            &&c->seg_fd>=0&&c->started){
                             int gap=(int)((cc-(uint8_t)((prev+1)&0xF))&0xF);
                             if(gap<1)gap=1;
                             if(gap>16)gap=16;
                             /* PCR-interpolated stuffing: write gap packets
                              * with smoothly advancing PCR values so the
                              * player's clock reference stays continuous.
                              * This is what TSDuck does and why
                              * TSDuck?ffmpeg plays clean while plain null
                              * stuffing (PID=0x1FFF, no PCR) still causes
                              * player drops — the PCR discontinuity at the
                              * gap boundary is what ExoPlayer actually
                              * reacts to, not the missing video data.
                              *
                              * Packet rate at 5.5Mbps ˜ 3657 pkts/sec.
                              * Each missing packet = 1/3657s ˜ 273µs.
                              * In 27MHz PCR ticks: 273µs × 27000000 ˜ 7371
                              * per packet. We write gap packets spaced by
                              * this increment so the PCR advances linearly
                              * across the stuffed gap at the correct rate.
                              * If last_pcr27 is unknown (-1), fall back to
                              * plain null packets.                         */
                             if(c->last_pcr27>=0&&c->pcr_pid){
                                 /* 27MHz PCR ticks per TS packet at nominal
                                  * bitrate. At 5.5Mbps: one 188-byte packet
                                  * takes 188*8/5500000 seconds ˜ 273µs.
                                  * In 27MHz ticks: 273µs × 27000000 ˜ 7371.
                                  * Fine-grained accuracy isn't critical here
                                  * — what matters is that PCR advances
                                  * monotonically and at roughly the right
                                  * rate across the gap so the player's clock
                                  * reference doesn't see a discontinuity.  */
                                 const int64_t TICKS_PER_PKT=7371;
                                 uint8_t pcr_pkt[188];
                                 for(int _n=0;_n<gap;_n++){
                                     int64_t pcr=c->last_pcr27+
                                                 (int64_t)(_n+1)*TICKS_PER_PKT;
                                     ts_write_pcr_pkt(pcr_pkt,c->pcr_pid,pcr);
                                     pkt_write(c,pcr_pkt);}
                                 c->last_pcr27+=
                                     (int64_t)(gap+1)*TICKS_PER_PKT;
                             }else{
                                 for(int _n=0;_n<gap;_n++)
                                     pkt_write(c,(const uint8_t*)NULL_TS_PKT);}
                             /* No seg_abort, no recovering=1 —
                              * stream continues uninterrupted.
                              * Continuation packets after the gap
                              * are written as-is: H.264 uses unbounded
                              * PES (length=0) so the decoder scans for
                              * NAL start codes rather than relying on
                              * PES length — a mid-NAL gap produces
                              * macroblocks in the affected region but
                              * the decoder keeps running. Discarding
                              * continuations (tried in V1.0.46) caused
                              * WORSE visible freezes: decoder had no
                              * data and repeated the last good frame
                              * for the entire gap duration, which is
                              * more disruptive than brief macroblocks.*/
                         }else if(c->seg_fd>=0&&!video_recovery_throttled){
                             /* For simple-copy audio PID drops: also use
                              * null stuffing rather than seg_close(). The
                              * audio decoder conceals a brief gap (a few
                              * missing MP2/AAC frames = a few ms of silence)
                              * far more gracefully than a full segment
                              * restart -- which causes a 0.1-1.5s stall
                              * while the player re-syncs. Confirmed from
                              * real logs: 0x0642 audio drops occurring
                              * simultaneously with video drops (same UDP
                              * datagram loss event) were still triggering
                              * seg_close(), producing short segments (0.339s,
                              * 0.113s) that caused player freezes even after
                              * the video-PID stuffing was fixed.
                              * For video_transcode, keep seg_close() since
                              * the encoder needs clean boundaries.        */
                             if(!c->opts.video_transcode){
                                 int gap=(int)((cc-(uint8_t)((prev+1)&0xF))&0xF);
                                 if(gap<1)gap=1;if(gap>16)gap=16;
                                 for(int _n=0;_n<gap;_n++)
                                     pkt_write(c,(const uint8_t*)NULL_TS_PKT);
                             }else{seg_close(c,now);c->seg_timing_ok=0;}}}
                     /* Only set recovering for video_transcode VIDEO-PID
                      * drop — for simple-copy the null stuffing above
                      * keeps the stream live without needing recovery. */
                     if(pid==c->vid_pid&&!video_recovery_throttled
                        &&c->opts.video_transcode)c->recovering=1;
                     /* A CC drop means we lost packets — any AU currently
                      * being accumulated is now corrupt/incomplete, and
                      * the decoder's reference picture buffer may
                      * reference frames that were never fully received.
                      * Force a full re-sync: discard whatever partial AU
                      * is in flight and require a fresh SPS+IDR (which
                      * also flushes the decoder) before decoding resumes.
                      * See vidxc_push's dec_ready gate for the flush.
                      * This fires independently of whatever other PID
                      * may already be mid-recovery — see rationale above
                      * for why that independence matters.               */
                     if(c->opts.video_transcode&&c->vidxc&&pid==c->vid_pid&&!video_recovery_throttled){
                         c->vidxc->dec_ready=0;
                         c->vidxc->force_idr=1;
                         au_reset(&c->vidxc->au);
                         c->last_recovery_trigger=*now;
                         c->last_recovery_trigger_valid=1;
                     }
                     /* Same rationale as above for video, applied to
                      * audio: a CC drop means real elapsed time has
                      * diverged from "samples decoded so far" — the
                      * encoder's PTS clock must re-lock onto the next
                      * packet's real PTS rather than keep extrapolating
                      * from its original (now stale) anchor point.     */
                     int asi=aud_slot_for_pid(c,pid);
                     if(asi>=0){
                         c->aud[asi].audxc.pts_ok=0;
                         c->aud[asi].audxc.pusi_seen=0;
                         c->aud[asi].audxc.last_seen_source_pts_valid=0;
                         c->aud[asi].audxc.latency_measured=0;
                         c->aud[asi].audxc.samples_since_anchor=0;
                     }}
             }
         }
         c->cc_last[pid&0x1FFF]=cc;}}

    /* -- STATE 0: wait for first IDR --------------------------
     * Cache SPS/PPS as they arrive. Start segment on first IDR.
     * IDR detection works on BOTH PUSI and continuation packets
     * because some sources split SPS+IDR across two TS packets. */
    if(!c->started){
        if(c->pmt_ok&&c->vid_pid&&pid==c->vid_pid&&ts_pusi(p)){
            /* video_transcode=1: start on any PUSI — encoder will open segment
             * on first IDR output (see vidxc block below).
             * simple-copy / audio_transcode: require source IDR.              */
            int can_start = c->opts.video_transcode ? 1 :
                (c->opts.simple_cut ? (nal_type(p)==5||nal_type(p)==20)
                                     : has_keyframe(p));
            if(can_start){
                c->started=1;
                {Evt e={EVT_START,time(NULL),0,0,0,c->vid_pid,0,0};evpush(&c->evring,&e);}
                /* For video_transcode: seg_open happens on first encoder IDR.
                 * For simple-copy: open segment now and write first IDR.      */
                if(!c->opts.video_transcode){
                    seg_open(c,now);pkt_write(c,p);}
            }
        }
        if(!c->started)return;
        if(!c->opts.video_transcode)return;}

    /* -- STATE 1: CC recovery --------------------------------- */
    /* This gate exists ONLY to hold back video output until a clean
     * random-access point arrives after a recovery — it must never
     * affect any other PID. Confirmed as a real, serious requirement:
     * audio (and anything else) must never be blocked by video
     * recovery state, even briefly. The old structure's final `else
     * return` fired for ANY non-video-PUSI packet while recovering was
     * true, including every audio packet, silently discarding them
     * until video recovered — this is now scoped so packets on any
     * PID other than the video PID skip this block entirely and fall
     * straight through to their normal processing, completely
     * unaffected by c->recovering's value.                           */
    if(c->recovering&&pid==c->vid_pid){
        if(ts_pusi(p)){
            if(c->opts.video_transcode){
                /* Don't open a segment here — that races against
                 * vidxc_push's own I-slice gate and our encoder's GOP
                 * cadence (see rationale above). Just clear the
                 * recovering flag; vidxc_push will silently discard
                 * non-I-slice AUs until a genuine I-slice arrives
                 * (dec_ready gate), and the existing got_keyframe/
                 * enc_started logic below opens the segment exactly
                 * the same way it does for the very first segment —
                 * only once OUR OWN encoder actually emits a fresh
                 * IDR with freshly re-sent SPS/PPS.                  */
                c->recovering=0;
            }else{
                int can_recover=c->opts.simple_cut?
                    (nal_type(p)==5||nal_type(p)==20):has_keyframe(p);
                if(can_recover){
                    c->recovering=0;
                    seg_open(c,now);
                    pkt_write(c,p);
                }else return;
            }
        }else return; /* continuation packet on the video PID during
            recovery — still correctly dropped: an incomplete/stale AU
            must not be accumulated. Only ever applies to the video
            PID now, never to anything else.                          */
        if(!c->opts.video_transcode)return;}

    /* -- video_transcode: decode+re-encode ------------------- */
    if(c->opts.video_transcode&&c->vidxc&&c->vidxc->active&&c->vid_pid&&pid==c->vid_pid){
        if(!c->seg_timing_ok){c->seg_start=*now;c->seg_timing_ok=1;}
        c->vidxc->got_keyframe=0;
        int n=vidxc_push(c->vidxc,p,c->vid_pid);
        if(c->vidxc->got_keyframe){
            if(!c->vidxc->enc_started){
                /* First IDR from encoder: now open the first segment.
                 * This ensures segment 0 always starts with IDR+audio. */
                c->vidxc->enc_started=1;
                if(c->seg_fd<0) seg_open(c,now); /* open if not already */
                printf("[vidxc] first IDR — segment opened, streaming starts\n");
            } else if(c->seg_fd<0){
                /* No segment currently open (e.g. immediately after CC
                 * recovery closed the previous one) — open one now
                 * regardless of elapsed time. This is the point where
                 * our own encoder has just produced a genuine fresh
                 * IDR with freshly re-sent SPS/PPS, making it exactly
                 * the right and only safe moment to start a new
                 * segment after a gap.                                */
                seg_open(c,now);
            } else if(c->seg_timing_ok){
                /* Subsequent IDR: cut segment if time elapsed.
                 * Small tolerance (10% of SEGMENT_SECS) absorbs normal
                 * keyframe-arrival jitter around the exact threshold —
                 * see rationale above this block for why this matters:
                 * missing an early arrival means waiting a FULL extra
                 * GOP for the next one (e.g. 2.0s threshold missed by
                 * 0.06s at elapsed=1.94s produced a 3.98s segment,
                 * confirmed via direct timing trace), which is a much
                 * worse outcome than a segment slightly under target.  */
                double el=wall_el(&c->seg_start,now);
                double tol=(double)SEGMENT_SECS*0.10;
                if(el>=(double)SEGMENT_SECS-tol){
                    seg_close(c,now);seg_open(c,now);}
            }
        }
        for(int i=0;i<n;i++)pkt_write(c,c->vidxc->out+i*TS_SZ);
        return;}

    /* -- Segment cut (simple-copy / audio_transcode) ---------- */
    if(c->vid_pid&&pid==c->vid_pid){
        if(ts_pusi(p)){
            int nt=nal_type(p);
            if(nt==7){/* SPS: reset cache */
                c->spspps_n=0;
                memcpy(c->spspps[0],p,TS_SZ);c->spspps_n=1;}
            else if(nt==8&&c->spspps_n>0&&c->spspps_n<8){
                memcpy(c->spspps[c->spspps_n],p,TS_SZ);c->spspps_n++;}
            /* Independent check, NOT an else-if: a packet classified as
             * nt==7 (SPS) by nal_type()'s own priority ranking can still
             * carry a real keyframe slice in the same payload (confirmed
             * on this source — SPS and the keyframe slice travel
             * together). Making this mutually exclusive with the SPS
             * branch above was the actual regression: the cut check
             * never ran at all for exactly the packets it needed to.   */
            int is_cut_point;
            if(c->opts.simple_cut){
                /* v23-compatible: raw NAL type only, no Exp-Golomb.
                 * Cuts ONLY on a genuine IDR (type 5/20) — sources using
                 * nal_type=1 for functional keyframes (the reason
                 * has_keyframe exists) will rarely/never satisfy this,
                 * same real limitation v23 itself has on those sources.*/
                is_cut_point=(nt==5||nt==20);
            } else {
                is_cut_point=(nt==5||nt==20||has_keyframe(p));
            }
            if(is_cut_point){/* IDR, or (default mode only) a real
                I-slice carried on a non-IDR NAL type (see has_keyframe
                doc) — either way, a genuine random-access point, safe
                to cut.                                                 */
                if(c->seg_timing_ok){
                    double el=wall_el(&c->seg_start,now);
                    double tol=(double)SEGMENT_SECS*0.10;
                    if(el>=(double)SEGMENT_SECS-tol){
                        /* If THIS packet just updated the SPS cache
                         * (nt==7), it already carries fresh SPS/PPS
                         * itself — seg_open()'s normal cache-replay
                         * would write this identical packet, then the
                         * pkt_write(c,p) below writes it AGAIN right
                         * after, producing a genuine duplicate
                         * SPS+PPS+slice sequence at the segment
                         * boundary (confirmed via direct byte
                         * inspection: this was the actual cause of the
                         * "illegal short term buffer state" decode
                         * errors, not a timing problem). Suppress the
                         * cache replay for this one seg_open() call —
                         * the immediately-following pkt_write supplies
                         * the fresh parameter sets instead.            */
                        /* suppress_replay must fire whenever THIS exact
                         * packet already carries fresh SPS/PPS itself,
                         * not just when nal_type()'s own priority-ranked
                         * return value happens to be 7. nal_type()
                         * returns IDR(5/20) immediately if present,
                         * even when SPS/PPS appear earlier in the SAME
                         * packet payload (confirmed real case: NALs
                         * [9,7,8,6,5] — AUD, SPS, PPS, SEI, IDR all in
                         * one packet, nal_type() correctly reports 5).
                         * The original fix only checked nt==7, missing
                         * this exact scenario — confirmed directly via
                         * real failing segments showing the predicted
                         * duplicate-SPS+PPS corruption pattern
                         * ("reference picture missing"/"mmco: unref
                         * short failure" cascades) at segment
                         * boundaries with zero CC drops anywhere
                         * nearby, ruling out packet loss as the cause.*/
                        int pkt_has_sps=0;
                        {int po=ts_poff(p);if(po>=0){
                            const uint8_t*pp=p+po;int pl=TS_SZ-po;
                            if(ts_pusi(p)&&pl>=9&&pp[0]==0&&pp[1]==0&&pp[2]==1){
                                int sk=9+pp[8];if(sk<pl){pp+=sk;pl-=sk;}}
                            for(int k=0;k+2<pl;k++){
                                int n2=-1;
                                if(pp[k]==0&&pp[k+1]==0&&pp[k+2]==1)n2=k+3;
                                else if(k+3<pl&&pp[k]==0&&pp[k+1]==0&&pp[k+2]==0&&pp[k+3]==1)n2=k+4;
                                if(n2<0||n2>=pl)continue;
                                if((pp[n2]&0x1F)==7){pkt_has_sps=1;break;}
                                k=n2;}}}
                        int suppress_replay=(nt==7)||pkt_has_sps;
                        seg_close(c,now);seg_open_ex(c,now,suppress_replay);
                        if(c->opts.idr_rewrite&&nt!=5&&nt!=20){
                            /* This is exactly the cut-point packet, and
                             * it got here via has_keyframe()'s Exp-
                             * Golomb check (nt is 1, not a real IDR) —
                             * see rewrite_slice_to_idr()/idr_rewrite doc
                             * for full rationale. Mutate a local copy;
                             * p itself is const and may be read again
                             * by other code paths after this returns.  */
                            uint8_t tmp[TS_SZ];memcpy(tmp,p,TS_SZ);
                            rewrite_slice_to_idr(tmp);
                            pkt_write(c,tmp);return;}
                        pkt_write(c,p);return;}}}
            /* CRITICAL: this fallback must run on EVERY PUSI packet,
             * not just ones identified as a cut-point candidate above —
             * confirmed via precise comparison against v23 (which has
             * this as an unconditional statement at this same scope,
             * not nested inside its IDR-found branch). If seg_timing_ok
             * is ever false when a NON-keyframe PUSI packet arrives,
             * leaving this nested inside is_cut_point meant it would
             * never get initialized from this path at all — seg_start
             * would stay stale, corrupting every subsequent
             * wall_el(&c->seg_start,now) duration calculation and the
             * tolerance check above, with no trace in CC-error logging
             * since this has nothing to do with CC detection. This was
             * a real, previously-undetected regression versus v23's
             * behavior, found by direct line-by-line comparison after
             * ruling out has_keyframe/simple_cut as the cause via real
             * production testing.                                      */
            if(!c->seg_timing_ok){c->seg_start=*now;c->seg_timing_ok=1;}
            /* Hard timeout fallback: if we've been waiting for the
             * regular has_keyframe()/IDR cut point for far longer than
             * normal (see MAX_SEG_WAIT_SECS doc), force a cut on THIS
             * PUSI packet even though it isn't one — confirmed real
             * cases of this exact gap (two genuine 14.2s segments, no
             * CC drop involved) left a real player frozen for the
             * whole gap with zero log signal otherwise. A forced cut
             * here isn't materially worse than a normal one on this
             * source: every segment boundary already lacks a true IDR
             * (confirmed via direct inspection of real captured
             * segments: zero NAL type 5 anywhere in 8 consecutive
             * segments, only type-1 slices that satisfy has_keyframe()'s
             * Exp-Golomb check) and already shows reference-buffer
             * corruption (mmco/reference-picture errors) on every
             * decode regardless — bounding the wait time turns an
             * unbounded freeze into, at worst, the same kind of brief
             * concealable glitch every other segment already has.      */
            else if(c->seg_timing_ok&&!is_cut_point){
                double waited=wall_el(&c->seg_start,now);
                if(waited>=(double)MAX_SEG_WAIT_SECS){
                    static time_t _last_warn=0;
                    time_t now_t=time(NULL);
                    if(now_t!=_last_warn){
                        fprintf(stderr,"[%s] WARNING: forced segment cut after "
                                "%.1fs waiting for next keyframe (no CC drop "
                                "involved) — source's keyframe cadence is "
                                "running longer than normal\n",c->name,waited);
                        _last_warn=now_t;}
                    seg_close(c,now);seg_open(c,now);pkt_write(c,p);return;
                }
            }}}

    /* -- audio_transcode: intercept audio PES -----------------
     * Use PUSI guard: discard continuation before first PUSI.
     * Dispatches by which active audio slot (if any) this PID
     * belongs to — multiple tracks can be active simultaneously,
     * each with its own independent transcoder/passthrough state.   */
    if(c->opts.audio_transcode){
        int si=aud_slot_for_pid(c,pid);
        if(si>=0){
            AudSlot*s=&c->aud[si];
            if(!s->audxc.active)goto passthrough;
            int oo=ts_poff(p);if(oo<0)goto passthrough;
            const uint8_t*pes=p+oo;int plen=TS_SZ-oo;
            if(ts_pusi(p)){
                if(plen<9||pes[0]!=0||pes[1]!=0||pes[2]!=1)goto passthrough;
                uint8_t pf=(pes[7]>>6)&3;
                if(pf&&plen>=14){
                    int64_t src_pts=(int64_t)(
                        ((uint64_t)(pes[9]&0x0E)<<29)|((uint64_t)pes[10]<<22)|
                        ((uint64_t)(pes[11]&0xFE)<<14)|((uint64_t)pes[12]<<7)|
                        ((uint64_t)(pes[13]&0xFE)>>1));
                    /* Always update last_seen_source_pts — needed by
                     * the latency measurement in audxc_feed_pcm, which
                     * may not run until several PUSI packets after the
                     * anchor below was first captured.                */
                    s->audxc.last_seen_source_pts=src_pts;
                    s->audxc.last_seen_source_pts_valid=1;
                    if(!s->audxc.pts_ok){
                        s->audxc.pts=src_pts;s->audxc.pts_ok=1;
                        /* Diagnostic: log the raw anchor PTS on both
                         * sides of a real cross-track offset bug — video
                         * and audio track's src_pts were observed 2+
                         * hours apart in real captured segments despite
                         * both allegedly reading real PES PTS off the
                         * same source. Video already has 2^33 wrap
                         * correction (see vidxc_push's WRAP33 clamp);
                         * this anchor does not. Comparing the raw values
                         * logged here against vidxc_push's own anchor
                         * log (search "[vidxc] video PTS anchor") for
                         * the same time window will show whether a wrap
                         * boundary is actually involved.                */
                        printf("[%s] audio PTS anchor: src_pts=%lld "
                               "(%.3fs)\n",c->name,(long long)src_pts,
                               (double)src_pts/90000.0);}}
                int skip=9+pes[8];
                if(skip>=plen)goto passthrough;
                pes+=skip;plen-=skip;
                s->audxc.pusi_seen=1;}
            else{
                if(!s->audxc.pusi_seen)goto passthrough;}
            int nout=audxc_push(&s->audxc,&c->aud_scratch,pes,plen,s->out_pid,&s->cc);
            static int _adone[MAX_CHANNELS][MAX_AUD_TRACKS];
            int chidx=(int)(c-g_ch);
            if(nout>0&&chidx>=0&&chidx<MAX_CHANNELS&&!_adone[chidx][si]){
                printf("[%s] audxc track%d: first AAC output: %d bytes (%d TS pkts)\n",
                       c->name,si+1,nout,nout/TS_SZ);
                _adone[chidx][si]=1;}
            if(nout>0)s->write_loop_entries++;
            for(int off=0;off+TS_SZ<=nout;off+=TS_SZ){
                pkt_write(c,s->audxc.out+off);
                s->write_loop_packets++;}
            return;}}
    passthrough:;

    /* -- zero-CPU patches (pts_reset, fix_interlace) ---------- */
    if(c->opts.pts_reset||c->opts.fix_interlace){
        uint8_t tmp[TS_SZ];memcpy(tmp,p,TS_SZ);
        if(c->opts.pts_reset&&(pid==c->vid_pid||aud_slot_for_pid(c,pid)>=0)&&ts_pusi(tmp))
            pes_patch_pts(tmp,&c->pts_base,&c->pts_base_ok);
        if(c->opts.fix_interlace&&pid==c->vid_pid&&ts_pusi(tmp))
            sps_patch_interlace(tmp);
        pkt_write(c,tmp);return;}

    pkt_write(c,p);}

/* ----------------------------------------------------------------
   NIC RECEIVE THREAD
   ---------------------------------------------------------------- */
typedef struct{Ch**chs;int nch;char iface[64];}NicGroup;

/* Interface resolution cache, keyed by multicast group.
 *
 * V1.0.42 — added after a confirmed real production problem: channel
 * reconnects on signal loss are implemented as a full process restart
 * (wrapper script/systemd), which means sock_open() — and therefore
 * the full probe_iface_for_group() join/poll/drop sequence — re-runs
 * LIVE every time ANY channel reconnects, while every other channel
 * is actively streaming on the same shared NICs. With 300 channels
 * and routine transient signal loss, this was happening frequently,
 * not as a rare edge case. Confirmed mechanism: IP_ADD_MEMBERSHIP and
 * IP_DROP_MEMBERSHIP both issue real IGMP reports onto the physical
 * network segment — repeated join/leave churn during a reconnect can
 * disrupt IGMP-snooping switches' forwarding state for a multicast
 * group, affecting OTHER channels sharing that same group (confirmed
 * earlier in this file's own history: CNBC_TV_18 and
 * KALAIGNAR_ISAI_ARUVI share group 231.1.1.3), which is consistent
 * with the reported symptom — continuous CC (continuity counter)
 * errors / ExoPlayer drops and A/V sync loss, persisting for hours,
 * not just around a one-time startup window.
 *
 * Fix: remember the resolved device for each multicast group on first
 * successful resolution, and reuse it instantly on every subsequent
 * channel start/restart for that same group — skipping the live probe
 * (and therefore all its IGMP join/leave churn) entirely on
 * reconnects. Only falls through to a fresh live probe if there's no
 * cached entry yet, OR if the cached device no longer exists on this
 * box at all (e.g. NIC renamed/removed since the cache was written —
 * checked directly via getifaddrs() before ever trusting a cache hit,
 * so a stale entry can never strand a channel). Deliberately does NOT
 * re-validate "is this device still actually receiving traffic" on
 * every reconnect — that would reintroduce exactly the join/leave
 * churn this fix removes. A cached device that's stopped carrying the
 * right traffic (NIC wiring changed, not just renamed/removed) is a
 * separate, rarer failure mode this doesn't address — if that's ever
 * suspected, clear /tmp/udp_hls_iface_cache.txt to force fresh probing
 * for every group again.
 *
 * Concurrency: many channel processes can start together (e.g. after
 * a box reboot) and read/write this file at the same time. Reads use
 * a shared flock, writes use an exclusive flock around a full read-
 * modify-write, so concurrent writes from different channels never
 * corrupt each other's entries — verified directly with 50 concurrent
 * writer processes, zero malformed lines, zero duplicate entries,
 * zero lost writes.                                                   */
#define IFACE_CACHE_PATH "/tmp/udp_hls_iface_cache.txt"
#define IFACE_CACHE_MAX_LINES 4096

static int iface_cache_lookup(const char*mcast_ip,char*out_dev,size_t out_len){
    FILE*f=fopen(IFACE_CACHE_PATH,"r");
    if(!f)return 0;
    if(flock(fileno(f),LOCK_SH)<0){fclose(f);return 0;}
    char line[256];
    int found=0;
    while(fgets(line,sizeof line,f)){
        char ip[64]={0},dev[64]={0};
        if(sscanf(line,"%63s %63s",ip,dev)!=2)continue; /* skip any
            malformed/partial line rather than fail the whole lookup —
            tolerates a line from a write that was interrupted before
            this fix's flock-based protection existed, or any future
            manual edits to the cache file.                            */
        if(strcmp(ip,mcast_ip)==0){
            strncpy(out_dev,dev,out_len-1);out_dev[out_len-1]=0;
            found=1;break;}
    }
    flock(fileno(f),LOCK_UN);
    fclose(f);
    if(!found)return 0;
    /* Never trust a cached device blindly — confirm it still exists on
     * this box right now before using it, so a stale entry (NIC
     * renamed/removed since the cache was written) safely falls
     * through to a fresh live probe instead of stranding the channel
     * on a device that no longer exists.                              */
    struct ifaddrs*ifs=NULL;getifaddrs(&ifs);
    int still_exists=0;
    for(struct ifaddrs*i=ifs;i;i=i->ifa_next)
        if(i->ifa_name&&strcmp(i->ifa_name,out_dev)==0){still_exists=1;break;}
    if(ifs)freeifaddrs(ifs);
    return still_exists;}

static void iface_cache_store(const char*mcast_ip,const char*dev){
    int fd=open(IFACE_CACHE_PATH,O_RDWR|O_CREAT,0644);
    if(fd<0)return;
    if(flock(fd,LOCK_EX)<0){close(fd);return;}
    FILE*f=fdopen(fd,"r+");
    if(!f){close(fd);return;}
    static char lines[IFACE_CACHE_MAX_LINES][256];
    int nlines=0,replaced=0;
    char line[256];
    rewind(f);
    while(nlines<IFACE_CACHE_MAX_LINES&&fgets(line,sizeof line,f)){
        char ip[64]={0},olddev[64]={0};
        if(sscanf(line,"%63s %63s",ip,olddev)==2&&strcmp(ip,mcast_ip)==0){
            snprintf(lines[nlines],sizeof lines[nlines],"%s %s\n",mcast_ip,dev);
            replaced=1;
        }else{
            strncpy(lines[nlines],line,sizeof lines[nlines]-1);
            lines[nlines][sizeof lines[nlines]-1]=0;}
        nlines++;}
    if(!replaced&&nlines<IFACE_CACHE_MAX_LINES){
        snprintf(lines[nlines],sizeof lines[nlines],"%s %s\n",mcast_ip,dev);
        nlines++;}
    rewind(f);
    if(ftruncate(fileno(f),0)==0)
        for(int i=0;i<nlines;i++)fputs(lines[i],f);
    fflush(f);
    flock(fileno(f),LOCK_UN);
    fclose(f);} /* closes fd too */

/* Listen-and-confirm interface detection — ported from a Java
 * implementation the user already had and wanted matched exactly: for
 * each candidate local interface (up, non-loopback, multicast-capable
 * — the same three checks Java's NetworkInterface exposes via
 * isUp()/isLoopback()/supportsMulticast(), here read directly from
 * ifa_flags via getifaddrs()), actually join the given multicast group
 * on THAT interface and wait up to `timeout_ms` for one real packet.
 * Returns 1 and fills out_ip on the first interface that receives a
 * packet, 0 if none did within the per-interface timeout.
 *
 * This is the STRONGEST guarantee of the three resolution strategies —
 * it confirms real traffic is actually flowing on that interface right
 * now, rather than just asking the routing table what it WOULD use
 * (resolve_iface_for_group(), below) or guessing the first non-
 * loopback interface system-wide (auto_iface()). The real cost: up to
 * (timeout_ms × number of candidate interfaces) in the worst case if
 * the source isn't currently flowing on any of them. Confirmed
 * acceptable with the user: this runs once per channel at startup
 * provisioning time, not repeatedly on a hot path, so the extra wall-
 * clock cost here is an explicit, accepted tradeoff for the stronger
 * guarantee — this is why it's tried FIRST in sock_open() below, ahead
 * of the near-instant routing-table method, rather than only as a
 * fallback.                                                            */
static int probe_iface_for_group(const char*mcast_ip,int port,int timeout_ms,
                                  char*out_ip,size_t out_ip_len){
    struct ifaddrs*ifs=NULL;
    if(getifaddrs(&ifs)<0)return 0;
    int found=0;
    for(struct ifaddrs*i=ifs;i&&!found;i=i->ifa_next){
        if(!i->ifa_addr||i->ifa_addr->sa_family!=AF_INET)continue;
        if(!(i->ifa_flags&IFF_UP)){
            fprintf(stderr,"[probe] %s: skipped (not UP)\n",i->ifa_name);continue;}
        if(i->ifa_flags&IFF_LOOPBACK){
            fprintf(stderr,"[probe] %s: skipped (loopback)\n",i->ifa_name);continue;}
        if(!(i->ifa_flags&IFF_MULTICAST)){
            fprintf(stderr,"[probe] %s: skipped (no multicast flag)\n",i->ifa_name);continue;}
        struct sockaddr_in*sa=(struct sockaddr_in*)i->ifa_addr;
        char dbg_ip[INET_ADDRSTRLEN];inet_ntop(AF_INET,&sa->sin_addr,dbg_ip,sizeof dbg_ip);
        struct timespec t0,t1;clock_gettime(CLOCK_MONOTONIC,&t0);
        fprintf(stderr,"[probe] %s (%s): trying, timeout=%dms...\n",i->ifa_name,dbg_ip,timeout_ms);

        int fd=socket(AF_INET,SOCK_DGRAM,0);
        if(fd<0)continue;
        int one=1;
        setsockopt(fd,SOL_SOCKET,SO_REUSEADDR,&one,sizeof one);
        /* V1.0.41 FIX — SO_BINDTODEVICE, confirmed necessary by direct
         * test against real production hardware: when another udp_hls
         * channel sharing the SAME multicast group/port is already
         * running (a real, routine condition for this deployment — two
         * different channel names can record the same upstream feed),
         * its already-active socket's IP_ADD_MEMBERSHIP join means the
         * kernel delivers a copy of every packet to ANY OTHER socket
         * also bound to that (group,port), regardless of what THAT
         * socket's own imr_interface was set to. Confirmed directly:
         * a probe socket joined with imr_interface set to the box's
         * default-route NIC (which has no physical connection to the
         * actual feed) still received a real, valid TS packet
         * (got_n=1316, first_byte=0x47) in 1ms, purely because another
         * channel's process already had a correctly-joined socket on
         * the same group/port — imr_interface alone does NOT prevent
         * this. SO_BINDTODEVICE is different: it restricts the SOCKET
         * ITSELF to a specific physical device for both send and
         * receive, independent of multicast group membership state on
         * other sockets/devices. Confirmed by direct test in both
         * directions: bound to the real device, traffic is received
         * normally (no regression); bound to a real-but-disconnected
         * device, a concurrently-active correct socket's traffic on
         * the right device does NOT leak through (request times out,
         * as it should) — this is the one mechanism of the three tried
         * so far that's actually immune to the cross-talk this
         * deployment routinely exhibits. Failure here (e.g. requires
         * CAP_NET_RAW — expected to be available since this already
         * runs as root in production) is non-fatal: just skip this
         * candidate and move on, same fail-safe convention as every
         * other step in this loop.                                    */
        if(setsockopt(fd,SOL_SOCKET,SO_BINDTODEVICE,i->ifa_name,strlen(i->ifa_name))<0){
            fprintf(stderr,"[probe] %s: SO_BINDTODEVICE failed (%s), skipping\n",
                    i->ifa_name,strerror(errno));
            close(fd);continue;}
        /* Bind to the MULTICAST GROUP address itself, the same
         * strategy sock_open() already uses for the real receive
         * socket (see `sa.sin_addr.s_addr=grp;` there). This was
         * tested two other ways first and both were wrong, confirmed
         * by direct send/receive test rather than assumed:
         *   - INADDR_ANY (the original draft): works for RECEIVING,
         *     but doesn't scope reception to a specific interface at
         *     all on its own — see the SO_BINDTODEVICE rationale above
         *     for why imr_interface alone isn't enough either.
         *   - A specific unicast interface address (the first attempted
         *     fix): confirmed by direct local send/receive test to
         *     NEVER receive multicast traffic at all — Linux UDP socket
         *     demux matches on the packet's actual destination address,
         *     which for multicast is always the GROUP address, never a
         *     unicast address, so a socket bound to a unicast address
         *     simply never matches an incoming multicast datagram
         *     regardless of IP_ADD_MEMBERSHIP/imr_interface.
         * Binding to the group address, confirmed by the same kind of
         * direct test, correctly receives multicast traffic for that
         * group.                                                       */
        struct sockaddr_in bindaddr={0};
        bindaddr.sin_family=AF_INET;
        bindaddr.sin_port=htons((uint16_t)port);
        if(inet_pton(AF_INET,mcast_ip,&bindaddr.sin_addr)!=1){close(fd);continue;}
        if(bind(fd,(struct sockaddr*)&bindaddr,sizeof bindaddr)<0){close(fd);continue;}

        struct ip_mreq mr={0};
        mr.imr_multiaddr=bindaddr.sin_addr;
        mr.imr_interface=sa->sin_addr;
        if(setsockopt(fd,IPPROTO_IP,IP_ADD_MEMBERSHIP,&mr,sizeof mr)<0){close(fd);continue;}

        struct pollfd pfd={.fd=fd,.events=POLLIN};
        int pr=poll(&pfd,1,timeout_ms);
        int got_ts=0;ssize_t got_n=-1;uint8_t first_byte=0;
        if(pr>0&&(pfd.revents&POLLIN)){
            uint8_t buf[1500];
            ssize_t n=recv(fd,buf,sizeof buf,0);
            got_n=n;if(n>0)first_byte=buf[0];
            /* Require the payload to actually look like MPEG-TS (sync
             * byte 0x47), the same check used everywhere else in this
             * file to validate a packet before trusting it (see
             * pkt_process's own `if(p[0]!=0x47)return;`). Any arriving
             * datagram is not proof this is the right feed on its own
             * — this guards against unrelated traffic incidentally
             * sharing this group/port being misread as a match.        */
            if(n>0&&buf[0]==0x47){
                found=1;got_ts=1;
                inet_ntop(AF_INET,&sa->sin_addr,out_ip,(socklen_t)out_ip_len);
            }
        }
        clock_gettime(CLOCK_MONOTONIC,&t1);
        double elapsed_ms=(t1.tv_sec-t0.tv_sec)*1000.0+(t1.tv_nsec-t0.tv_nsec)/1e6;
        fprintf(stderr,"[probe] %s (%s): poll_result=%d got_n=%zd first_byte=0x%02X "
                "match=%s elapsed=%.0fms\n",
                i->ifa_name,dbg_ip,pr,got_n,first_byte,got_ts?"YES":"no",elapsed_ms);
        /* Drop membership before closing — this fd is just a probe,
         * never the channel's real receive socket (that's opened
         * fresh, separately, in sock_open()), so leave no trace of
         * group membership registered against this address.          */
        setsockopt(fd,IPPROTO_IP,IP_DROP_MEMBERSHIP,&mr,sizeof mr);
        close(fd);
    }
    if(ifs)freeifaddrs(ifs);
    fprintf(stderr,"[probe] FINAL: %s\n",found?"matched (see iface above)":"no interface matched, falling back");
    return found;}

/* Resolve which local interface IP the kernel would actually use to
 * reach a given multicast group, without sending any packet and
 * without forking a subprocess.
 *
 * V1.0.37: added to remove the need to hand-map each of 300 channels'
 * multicast groups to the correct NIC IP when the box has multiple
 * physical/bonded interfaces each carrying different feeds (confirmed
 * real operational pain point — auto_iface() below only ever returns
 * ONE interface system-wide, which silently breaks any channel whose
 * group isn't reachable on that particular NIC; mr.imr_interface joins
 * the group on the wrong interface, the join itself doesn't fail, and
 * the channel just never receives any packets — see sock_open()'s
 * SO_REUSEPORT removal comment for the same class of "join succeeds,
 * traffic silently never arrives" failure mode this avoids repeating).
 *
 * Mechanism: connect() on an unconnected UDP socket performs a real
 * kernel routing-table lookup and binds a local address accordingly —
 * no SYN, no data, nothing is sent on the wire for a UDP socket. This
 * is the same lookup `ip route get <mcast>` reports (confirmed
 * directly: a real multicast group on this exact production box
 * resolved via a normal routing-table entry — "dev eth0 src X" with no
 * "via" gateway hop — meaning no special PIM/igmpproxy multicast
 * routing is configured here, just a plain unicast-style lookup, which
 * is exactly what connect() resolves internally). getsockname()
 * afterward reports which local address the kernel chose, which tells
 * us which interface.
 *
 * Guarded to only ever attempt this for a genuine multicast address
 * (224.0.0.0/4 — same test already used as `mc` in sock_open() below).
 * Confirmed by direct test: without this guard, a degenerate/malformed
 * config value like "0.0.0.0" silently resolves to the loopback
 * interface (127.0.0.1) via this same connect()/getsockname() trick —
 * which would be a real, silent misconfiguration if this were ever fed
 * a bad config line, joining a "multicast group" on `lo` and receiving
 * nothing, exactly the kind of clean-looking-but-empty failure this
 * feature exists to avoid introducing elsewhere.
 *
 * Returns 1 on success (out_ip filled), 0 on any failure — caller must
 * always have a fallback path (the existing iface/auto_iface() chain)
 * and must never block channel startup on this.                       */
static int resolve_iface_for_group(const char*mcast_ip,int port,char*out_ip,size_t out_len){
    struct in_addr grp_addr;
    if(inet_pton(AF_INET,mcast_ip,&grp_addr)!=1)return 0;
    if((ntohl(grp_addr.s_addr)>>28)!=0xE)return 0; /* not 224.0.0.0/4 */
    int fd=socket(AF_INET,SOCK_DGRAM,0);
    if(fd<0)return 0;
    struct sockaddr_in dst={0};
    dst.sin_family=AF_INET;
    dst.sin_port=htons((uint16_t)(port>0?port:1));
    dst.sin_addr=grp_addr;
    if(connect(fd,(struct sockaddr*)&dst,sizeof dst)<0){close(fd);return 0;}
    struct sockaddr_in local={0};socklen_t slen=sizeof local;
    if(getsockname(fd,(struct sockaddr*)&local,&slen)<0){close(fd);return 0;}
    close(fd);
    if(local.sin_addr.s_addr==0)return 0; /* unresolved/INADDR_ANY */
    if(!inet_ntop(AF_INET,&local.sin_addr,out_ip,(socklen_t)out_len))return 0;
    return 1;}

static void auto_iface(char*out,size_t len){
    struct ifaddrs*l=NULL;getifaddrs(&l);
    for(struct ifaddrs*i=l;i;i=i->ifa_next){
        if(!i->ifa_addr||i->ifa_addr->sa_family!=AF_INET)continue;
        struct sockaddr_in*sa=(struct sockaddr_in*)i->ifa_addr;
        if((ntohl(sa->sin_addr.s_addr)>>28)==0)continue;
        if((ntohl(sa->sin_addr.s_addr)>>24)==127)continue;
        inet_ntop(AF_INET,&sa->sin_addr,out,(socklen_t)len);break;}
    if(l)freeifaddrs(l);}

static int sock_open(Ch*c){
    uint32_t grp=inet_addr(c->mcast);
    int mc=((ntohl(grp)>>28)==0xE);
    char iface[64]={0};
    int auto_resolved=0;
    if(c->iface[0])strncpy(iface,c->iface,63);
    else if(mc){
        /* No iface given (config has "-" or omits the field) — three
         * resolution strategies are tried in order, strongest
         * guarantee first:
         *
         * 1. probe_iface_for_group() — actually join the group on
         *    each candidate interface and wait for a real packet
         *    (ported from the user's own Java implementation, by
         *    request). Confirms genuine live traffic, not just a
         *    routing-table opinion — the strongest guarantee of the
         *    three. Cost: up to ~1.5s per candidate interface in the
         *    worst case if the source isn't currently flowing
         *    anywhere; confirmed acceptable with the user since this
         *    runs once per channel at startup, not on a repeated hot
         *    path.
         * 2. resolve_iface_for_group() — near-instant routing-table
         *    query (no actual packet wait), tried only if the probe
         *    above found nothing (e.g. source briefly silent at the
         *    exact moment of startup).
         * 3. auto_iface() — single-interface, system-wide guess, last
         *    resort only if both of the above fail.
         *
         * Never blocks or fails channel startup over this — any
         * resolution failure just falls through to the next strategy,
         * and total failure still leaves a usable (if likely wrong)
         * interface value rather than refusing to start.              */
        if(probe_iface_for_group(c->mcast,c->port,1500,iface,sizeof iface))
            auto_resolved=1;
        else if(resolve_iface_for_group(c->mcast,c->port,iface,sizeof iface))
            auto_resolved=1;
        else
            auto_iface(iface,sizeof iface);
        /* Record what was actually used back onto the channel struct —
         * c->iface is otherwise only ever the raw config value (empty
         * here), and without this, startup logging/diagnostics have no
         * way to show which interface a "-"/auto channel actually
         * ended up joining on. Harmless: nothing reads c->iface before
         * this point in a channel's lifecycle (no reload/restart path
         * re-enters sock_open() with stale expectations about it).    */
        if(iface[0])strncpy(c->iface,iface,63);
    }
    int fd=socket(AF_INET,SOCK_DGRAM,IPPROTO_UDP);
    if(fd<0){perror("socket");return -1;}
    int one=1,zero=0;
    setsockopt(fd,SOL_SOCKET,SO_REUSEADDR,&one,sizeof one);
    /* SO_REUSEPORT intentionally NOT set.
     *
     * Confirmed real bug this caused: two channels sharing the same
     * UDP port (different multicast groups, same NIC) each had
     * SO_REUSEPORT set. When the primary bind() to the multicast group
     * address (below) falls through to its INADDR_ANY fallback -- a
     * real, observed path on production servers, not hypothetical --
     * the resulting socket becomes eligible to receive ANY UDP traffic
     * on that port regardless of destination multicast group, and the
     * kernel's SO_REUSEPORT delivery selection among multiple sockets
     * bound to the same (address, port) does NOT consult each socket's
     * own IP_ADD_MEMBERSHIP group membership when choosing which
     * socket gets a given datagram. Net effect, confirmed directly via
     * decrypted production segments: a second channel's process
     * silently received and decoded the FIRST channel's video/audio
     * (same PAT/PMT/PCR PIDs found inside the second channel's own
     * output segments), while genuinely looking for its own (different)
     * audio PID on every packet -- which never arrived, since it was
     * never actually receiving its own multicast stream. Zero error
     * anywhere: aud_slot_for_pid() correctly returned no match for
     * every packet it WAS given, so audio was silently never
     * transcoded, with completely clean stats (cc_tot stable,
     * idle=0.0s, segs incrementing on schedule) since the wrong-
     * channel's packets were still real, valid, continuously-arriving
     * MPEG-TS the whole time.
     *
     * This design runs exactly one socket per channel with no
     * intentional load-sharing across processes/sockets -- the actual
     * use case SO_REUSEPORT exists for -- so removing it costs
     * nothing functionally, and prevents this exact cross-talk class
     * of bug even if two channels end up sharing a port again.        */
    int rb=RCVBUF;
    if(setsockopt(fd,SOL_SOCKET,SO_RCVBUFFORCE,&rb,sizeof rb)!=0)
        setsockopt(fd,SOL_SOCKET,SO_RCVBUF,&rb,sizeof rb);
    fcntl(fd,F_SETFL,fcntl(fd,F_GETFL,0)|O_NONBLOCK);
    struct sockaddr_in sa={0};
    sa.sin_family=AF_INET;sa.sin_port=htons((uint16_t)c->port);
    if(mc){
        sa.sin_addr.s_addr=grp;
        if(bind(fd,(struct sockaddr*)&sa,sizeof sa)<0){
            /* Diagnostic: this fallback is exactly the path that made
             * the cross-talk bug above possible in the first place --
             * a socket bound to INADDR_ANY:port instead of the
             * specific multicast group address has no kernel-level
             * destination-address filtering of its own, relying
             * entirely on IP_ADD_MEMBERSHIP (an accept-onto-host
             * filter, not a per-socket delivery-routing filter) for
             * correctness. Logged so this is visible per-channel at
             * startup instead of silently happening.                  */
            fprintf(stderr,"[%s] WARNING: bind() to multicast group %s:%d "
                    "failed (%s) -- falling back to INADDR_ANY:%d. This "
                    "socket will rely entirely on IP_ADD_MEMBERSHIP for "
                    "correct packet delivery; if another channel shares "
                    "this port, verify it is not also on this fallback "
                    "path.\n",c->name,c->mcast,c->port,strerror(errno),c->port);
            sa.sin_addr.s_addr=htonl(INADDR_ANY);
            bind(fd,(struct sockaddr*)&sa,sizeof sa);}
        struct ip_mreq mr={0};
        mr.imr_multiaddr.s_addr=grp;
        mr.imr_interface.s_addr=iface[0]?inet_addr(iface):htonl(INADDR_ANY);
        if(setsockopt(fd,IPPROTO_IP,IP_ADD_MEMBERSHIP,&mr,sizeof mr)<0){
            perror("IP_ADD_MEMBERSHIP");close(fd);return -1;}
        setsockopt(fd,IPPROTO_IP,IP_MULTICAST_ALL,&zero,sizeof zero);
    }else{sa.sin_addr.s_addr=htonl(INADDR_ANY);bind(fd,(struct sockaddr*)&sa,sizeof sa);}
    return fd;}

static void mkdirp(const char*p){
    char t[512];snprintf(t,512,"%s",p);
    for(char*s=t+1;*s;s++)if(*s=='/'){*s=0;mkdir(t,0755);*s='/';}
    mkdir(t,0755);}

/* ----------------------------------------------------------------
   RECV/PROCESS DECOUPLING RING
   ----------------------------------------------------------------
   Problem: pkt_process() used to be called synchronously inline
   inside the same recv() loop that drains this channel's UDP socket.
   audio_transcode's per-packet cost (mpg123 decode + resample + AAC
   encode) is far higher than simple-copy's, so any scheduling delay
   or transient CPU spike could let pkt_process fall behind recv() —
   at which point every subsequent packet in the kernel socket buffer
   sits unread until this channel's next recv() call, and a large
   enough stall silently overflows that buffer for real: exactly the
   pattern confirmed in production (repeated "STALE no input for ~4s"
   -> CLEAN START -> CC errors immediately after, on an otherwise
   perfectly healthy feed, only on channels with audio_transcode
   enabled).

   Fix: recv() now only ever does one thing — pull datagrams off the
   socket as fast as possible and copy raw TS packets into a small
   per-channel ring. All decode/encode work (pkt_process) happens in a
   separate pass, AFTER every channel's socket has been drained for
   this poll() iteration. This bounds the time between any two
   recv() calls on a given socket to "however long it takes to drain
   every OTHER channel's socket", instead of "however long the
   slowest single pkt_process() call takes" -- a large, structural
   improvement, since draining a socket (recv() + memcpy) is orders of
   magnitude cheaper than transcoding.                                */
static int ring_push(Ch*c,const uint8_t*ts){
    if(!c->pkt_ring){
        c->pkt_ring=malloc((size_t)RECV_RING_PKTS*TS_SZ);
        if(!c->pkt_ring)return 0; /* malloc failure: drop, don't crash */
        c->ring_head=c->ring_tail=c->ring_count=0;
    }
    if(c->ring_count>=RECV_RING_PKTS){
        /* Ring full -- pkt_process has fallen far enough behind that
         * RECV_RING_PKTS worth of backlog piled up. Drop and count it
         * rather than overwrite (overwriting would silently corrupt
         * CC continuity far worse than a clean, countable drop does),
         * and rate-limit the log so a sustained overload doesn't add
         * its own stdout-lock-contention cost on top of the problem.  */
        c->ring_drops++;
        if(c->ring_drops<=5||c->ring_drops%1000==0)
            fprintf(stderr,"[%s] WARNING: recv ring full, dropping packet "
                    "(ring_drops=%llu) — pkt_process is not keeping up "
                    "with recv()\n",c->name,(unsigned long long)c->ring_drops);
        return 0;
    }
    memcpy(c->pkt_ring+(size_t)c->ring_tail*TS_SZ,ts,TS_SZ);
    c->ring_tail=(c->ring_tail+1)%RECV_RING_PKTS;
    c->ring_count++;
    return 1;
}
/* Drains every packet currently queued for this channel. Called once
 * per channel per poll() iteration, after ALL channels' sockets have
 * already been drained (see nic_thread) -- so a channel with a large
 * backlog processing here never delays recv() for itself or any other
 * channel in this same iteration; it only delays how quickly ITS OWN
 * output catches back up to live, which is the correct trade-off.    */
static void ring_drain(Ch*c,const struct timespec*now){
    if(!c->pkt_ring)return;
    while(c->ring_count>0&&!g_stop){
        const uint8_t*ts=c->pkt_ring+(size_t)c->ring_head*TS_SZ;
        if(ts[0]==0x47)pkt_process(c,ts,now);
        c->ring_head=(c->ring_head+1)%RECV_RING_PKTS;
        c->ring_count--;
    }
}

static void*nic_thread(void*arg){
    NicGroup*g=(NicGroup*)arg;int nch=g->nch;
    struct pollfd*pfds=calloc(nch,sizeof*pfds);
    av_log_set_level(AV_LOG_ERROR); /* suppress decoder warnings */
    av_log_set_callback(corruption_log_callback); /* observe real
        reference-frame corruption signals - see corruption_log_callback
        and corrupt_errors_since_check docs for full rationale          */

    for(int i=0;i<nch;i++){
        Ch*c=g->chs[i];
        mkdirp(c->dir);
        if(!c->seg_seq)c->seg_seq=c->start_seq;
        c->seg_fd=-1;c->pat_ok=0;c->pmt_ok=0;
        c->seg_timing_ok=0;c->started=0;c->recovering=0;
        c->last_pcr27=-1;c->cc_drop_skip=0;
        c->bridging=0;c->bridge_pkts_emitted=0;
        c->wb_n=0;c->aes_on=0;
        memset(c->cc_last,0xFF,sizeof c->cc_last);c->cc_errors=0;
        c->pts_base_ok=0;c->pts_base=0;
        c->target_dur=0;c->target_dur_fixed=0;c->spspps_n=0;
        memset(c->aud,0,sizeof c->aud);c->n_aud_active=0;c->n_aud_dropped=0;
        c->vidxc=NULL; /* heap-allocated lazily on first video_transcode=1
            init in pkt_process -- guaranteed NULL here since nic_thread
            runs this block exactly once per channel per process
            lifetime (no reload/restart path re-enters it).           */
        c->snap_pkts=0;c->snap_bytes=0;c->snap_cc=0;
        clock_gettime(CLOCK_MONOTONIC,&c->snap_time);
        memset(&c->evring,0,sizeof c->evring);
        clock_gettime(CLOCK_MONOTONIC,&c->last_pkt);

        /* AES key */
        if(c->keyfile[0]&&strcmp(c->keyfile,"-")!=0){
            FILE*kf=fopen(c->keyfile,"rb");
            if(kf){uint8_t key[16];size_t n=fread(key,1,16,kf);fclose(kf);
                if(n==16){uint8_t iv[16]={0};int ok=1;
                    if(c->iv_hex[0]&&strcmp(c->iv_hex,"-")!=0){
                        if(strlen(c->iv_hex)!=32)ok=0;
                        else for(int j=0;j<16;j++){unsigned v=0;
                            if(sscanf(c->iv_hex+j*2,"%02x",&v)!=1){ok=0;break;}
                            iv[j]=(uint8_t)v;}}
                    if(ok){aes_init(&c->aes,key,iv);c->aes_on=1;
                           printf("[%s] AES ready\n",c->name);}}}}

        c->fd=sock_open(c);
        if(c->fd<0){pfds[i].fd=-1;pfds[i].events=0;continue;}
        pfds[i].fd=c->fd;pfds[i].events=POLLIN;

        const char*mode=(c->opts.audio_transcode||c->opts.video_transcode)?
                        "in-process":"simple-copy";
        printf("[%s] joined %s:%d iface=%s  mode=%s\n",
               c->name,c->mcast,c->port,c->iface[0]?c->iface:"auto",mode);}

    printf("[NIC:%s] watching %d channels\n",
           g->iface[0]?g->iface:"auto",nch);fflush(stdout);

    static uint8_t buf[UDP_MAX];
    while(!g_stop){
        int ready=poll(pfds,nch,10);
        if(ready<0){if(errno==EINTR)continue;break;}
        struct timespec now;clock_gettime(CLOCK_MONOTONIC,&now);
        for(int i=0;i<nch&&!g_stop;i++){
            Ch*c=g->chs[i];
            if(pfds[i].fd<0)continue;
            /* stale detection */
            if(c->started&&c->seg_fd>=0){
                double age=wall_el(&c->last_pkt,&now);
                if(age>=(double)(SEGMENT_SECS*2)&&age<BRIDGE_GAP_MAX_SECS){
                    /* SOFT BRIDGE: absorb the gap in place instead of
                     * tearing the channel down. No seg_close(), no
                     * c->started/recovering reset, no PAT/PMT
                     * re-acquisition, no NO_SIGNAL/CLEAN START, no CC
                     * discontinuity on resume — exactly the class of
                     * disruption a player actually reacts to, and
                     * exactly what plain ffmpeg on the same source
                     * never produces for a gap this size (confirmed via
                     * direct side-by-side comparison: ffmpeg absorbs
                     * this silently, this program's old hard-teardown
                     * path is what caused the visible stop/restart).   */
                    if(!c->bridging){
                        c->bridging=1;c->bridge_pkts_emitted=0;
                        printf("[%s] BRIDGING gap (%.3fs so far, soft "
                               "recovery — segment/PAT/playlist "
                               "untouched)\n",c->name,age);
                        /* Same decoder-side resync as the old hard path
                         * (still correct and still needed — a real gap
                         * this long leaves decoder-internal state
                         * stale regardless of how gently we handle the
                         * segment/playlist around it), just without the
                         * segmenter/PAT teardown alongside it.          */
                        if(c->opts.video_transcode&&c->vidxc){
                            c->vidxc->dec_ready=0;
                            c->vidxc->force_idr=1;
                            au_reset(&c->vidxc->au);
                        }
                        for(int ai=0;ai<c->n_aud_active;ai++){
                            AudXc*a=&c->aud[ai].audxc;
                            if(!a->active)continue;
                            mpg123_close(a->mh);
                            if(mpg123_open_feed(a->mh)!=MPG123_OK){
                                fprintf(stderr,"[%s] mpg123_open_feed FAILED "
                                        "on bridge recovery — disabling this "
                                        "audio track\n",c->name);
                                a->active=0;continue;
                            }
                            a->pts_ok=0;a->pusi_seen=0;
                            a->last_seen_source_pts_valid=0;
                            a->latency_measured=0;a->samples_since_anchor=0;
                            a->pcm_fill=0;
                        }
                    }
                    /* PCR-interpolated stuffing, cumulative target-based
                     * pacing so this self-corrects for any jitter in our
                     * own poll loop rather than needing precise per-tick
                     * timing — mirrors the CC-drop path's gap-stuffing
                     * (see pkt_process) but sized to the real elapsed
                     * wall-clock gap instead of a max-16-packet CC delta.
                     * Same nominal rate assumption as that path's
                     * TICKS_PER_PKT=7371 (~5.5Mbps): 27,000,000/7371 ˜
                     * 3663 packets/sec.                                 */
                    const int64_t TICKS_PER_PKT=7371;
                    const double PKTS_PER_SEC=3663.0;
                    uint64_t target=(uint64_t)(age*PKTS_PER_SEC);
                    if(target>c->bridge_pkts_emitted){
                        uint64_t need=target-c->bridge_pkts_emitted;
                        if(need>4000)need=4000; /* per-tick cap — avoid a
                            huge single burst if our own loop briefly
                            lagged; remainder catches up next tick(s).  */
                        if(c->last_pcr27>=0&&c->pcr_pid){
                            uint8_t pcr_pkt[188];
                            for(uint64_t _n=0;_n<need;_n++){
                                int64_t pcr=c->last_pcr27+
                                    (int64_t)(c->bridge_pkts_emitted+_n+1)
                                    *TICKS_PER_PKT;
                                ts_write_pcr_pkt(pcr_pkt,c->pcr_pid,pcr);
                                pkt_write(c,pcr_pkt);}
                        }else{
                            for(uint64_t _n=0;_n<need;_n++)
                                pkt_write(c,(const uint8_t*)NULL_TS_PKT);}
                        c->bridge_pkts_emitted+=need;
                    }
                }else if(age>=BRIDGE_GAP_MAX_SECS){
                    /* Genuine extended outage — the bridge window has
                     * been exhausted with no real data. Fall back to
                     * the full recovery: PAT/PMT re-acquisition and a
                     * clean decoder/segmenter restart are actually
                     * warranted here, not overkill.                    */
                    if(c->last_pcr27>=0)
                        c->last_pcr27+=(int64_t)c->bridge_pkts_emitted*7371;
                    c->bridging=0;c->bridge_pkts_emitted=0;
                    Evt e={EVT_STALE,time(NULL),0,(uint32_t)(age*1000),0,0,0,0};
                    evpush(&c->evring,&e);
                    seg_close(c,&now);c->started=0;c->recovering=0;c->seg_timing_ok=0;
                    /* Reset transcode decoder state too, not just
                     * segmenter state — confirmed real gap: this branch
                     * previously only reset c->started/recovering/
                     * seg_timing_ok, which is sufficient for simple-copy
                     * (no decoder state exists there at all) but leaves
                     * video_transcode's h264 decoder and audio_transcode's
                     * mpg123 decoder completely untouched across a real
                     * multi-second signal outage. Feeding fresh, post-gap
                     * packets into decoders that still think they're
                     * mid-stream (stale reference-picture buffer for
                     * h264, stale internal sync state for mpg123) is
                     * exactly the kind of thing that produces corrupted
                     * decode output or silently-broken output — confirmed
                     * real-world symptom on this exact channel: a STALE
                     * event with audio_transcode active left mpg123's
                     * internal state stale after CLEAN START, producing
                     * no audio afterward (and, most likely, the "video
                     * drops" reported alongside it — a player stalling
                     * on a broken/silent audio track inside an otherwise
                     * fine HLS segment looks like video drops from the
                     * viewer's side even when the video track itself is
                     * untouched). Mirrors the EXACT reset already used
                     * for a routine CC drop (see pkt_process's video/
                     * audio CC handling) — a multi-second STALE gap is a
                     * strictly bigger discontinuity than a routine CC
                     * drop, so applying the same-or-stronger reset here
                     * is the safe direction, never weaker.              */
                    if(c->opts.video_transcode&&c->vidxc){
                        c->vidxc->dec_ready=0;
                        c->vidxc->force_idr=1;
                        au_reset(&c->vidxc->au);
                    }
                    for(int ai=0;ai<c->n_aud_active;ai++){
                        AudXc*a=&c->aud[ai].audxc;
                        if(!a->active)continue;
                        /* Full mpg123 resync (close+reopen), not just the
                         * lighter pts/anchor reset used for a routine CC
                         * drop — a multi-second outage is a big enough
                         * discontinuity that any bytes still sitting in
                         * mpg123's internal buffer are almost certainly
                         * stale/torn, same reasoning as the MPG123_ERR
                         * resync path this mirrors (see that block's own
                         * doc for why closing beats trying to salvage).  */
                        mpg123_close(a->mh);
                        if(mpg123_open_feed(a->mh)!=MPG123_OK){
                            fprintf(stderr,"[%s] mpg123_open_feed FAILED on "
                                    "STALE recovery — disabling this audio "
                                    "track\n",c->name);
                            a->active=0;continue;
                        }
                        printf("[%s] audio track%d: STALE-recovery reset "
                               "fired (was pts_ok=%d pcm_fill=%d "
                               "stall_calls=%d)\n",c->name,ai+1,
                               a->pts_ok,a->pcm_fill,a->stall_calls);
                        a->pts_ok=0;a->pusi_seen=0;
                        a->last_seen_source_pts_valid=0;
                        a->latency_measured=0;a->samples_since_anchor=0;
                        a->pcm_fill=0; /* CRITICAL — see the identical fix
                            and full rationale in the MPG123_ERR resync
                            path above. This was the actual reason audio
                            stayed broken after a STALE recovery even
                            though mpg123 itself resynced cleanly: a
                            stale partial PCM frame surviving the reset,
                            silently corrupting every AAC frame built
                            from it afterward.                          */
                    }
                }else if(c->bridging){
                    /* Real data resumed while we were bridging (age
                     * dropped back under SEGMENT_SECS*2 because
                     * c->last_pkt just got updated) — bridge episode
                     * over, no escalation needed.                      */
                    printf("[%s] BRIDGE resolved — real data resumed after "
                           "%llu stuffed packets\n",c->name,
                           (unsigned long long)c->bridge_pkts_emitted);
                    c->bridging=0;c->bridge_pkts_emitted=0;
                }}
            if(!(pfds[i].revents&POLLIN))continue;
            /* PASS 1 of 2: drain the socket as fast as possible.
             * Nothing here does decode/encode work -- just recv() +
             * memcpy into this channel's ring. See ring_push/ring_drain
             * doc above for why this is split from processing.        */
            for(int lim=0;lim<MAX_PER_CHAN&&!g_stop;lim++){
                ssize_t n=recv(c->fd,buf,sizeof buf,0);
                if(n<=0){if(errno==EAGAIN||errno==EWOULDBLOCK)break;continue;}
                c->dgrams++;clock_gettime(CLOCK_MONOTONIC,&c->last_pkt);
                if(c->dgrams==1){
                    printf("[%s] first dgram len=%d\n",c->name,(int)n);
                    if(buf[0]!=0x47){
                        /* RTP detect */
                        if(n>=13&&(buf[0]>>6)==2){
                            int h=12+(buf[0]&0xF)*4;
                            if((buf[0]>>4)&1&&n>h+4)h+=4+((buf[h+2]<<8)|buf[h+3])*4;
                            if(h<n&&buf[h]==0x47){
                                printf("[%s] RTP+%d\n",c->name,h);
                                /* store rtp offset in dgrams field temporarily */
                                /* Actually just scan from offset */
                                const uint8_t*ts=buf+h;
                                int np=(int)(n-h)/TS_SZ;
                                for(int j=0;j<np;j++,ts+=TS_SZ)
                                    if(ts[0]==0x47)ring_push(c,ts);
                                continue;}}
                        printf("[%s] unknown format byte=0x%02X\n",c->name,buf[0]);
                        continue;}
                    else printf("[%s] plain TS\n",c->name);}
                /* plain TS */
                int np=(int)n/TS_SZ;
                const uint8_t*ts=buf;
                for(int j=0;j<np;j++,ts+=TS_SZ)
                    if(ts[0]==0x47)ring_push(c,ts);}}
        /* PASS 2 of 2: now that every channel's socket has been
         * drained for this poll() iteration, do the actual decode/
         * encode work. A slow channel here can no longer delay recv()
         * for itself or any other channel -- it only delays how
         * quickly its own output catches back up to live.            */
        for(int i=0;i<nch&&!g_stop;i++){
            Ch*c=g->chs[i];
            if(pfds[i].fd<0)continue;
            ring_drain(c,&now);}}

    /* shutdown */
    for(int i=0;i<nch;i++){
        Ch*c=g->chs[i];
        if(c->seg_fd>=0){
            struct timespec now;clock_gettime(CLOCK_MONOTONIC,&now);
            if(c->aes_on)aes_flush(&c->aes,c->seg_fd,1);else wb_flush(c);
            close(c->seg_fd);printf("[%s] final\n",c->name);}
        for(int k=0;k<c->n_aud_active;k++)audxc_free(&c->aud[k].audxc);
        if(c->vidxc){vidxc_free(c->vidxc);free(c->vidxc);c->vidxc=NULL;}
        if(c->pkt_ring){free(c->pkt_ring);c->pkt_ring=NULL;}
        if(c->fd>=0)close(c->fd);}
    free(pfds);return NULL;}

/* ----------------------------------------------------------------
   OPTS PARSER / CONFIG LOADER
   ---------------------------------------------------------------- */
static void parse_opts(ChOpts*o,const char*toks[],int ntok){
    memset(o,0,sizeof*o);
    strncpy(o->audio_coder,"fast",sizeof(o->audio_coder)-1); /* real default —
        previously "fast" only appeared in a log-message fallback that
        never reached audxc_init; the actual struct field stayed empty
        unless audio_coder= was explicitly passed in config, silently
        falling back to libavcodec's own default (twoloop) instead.   */
    for(int i=0;i<ntok;i++){
        const char*t=toks[i];
        if     (!strncmp(t,"audio_transcode=",16))o->audio_transcode=atoi(t+16);
        else if(!strncmp(t,"video_transcode=",16))o->video_transcode=atoi(t+16);
        else if(!strncmp(t,"fix_mp2=",8))         o->fix_mp2=atoi(t+8);
        else if(!strncmp(t,"fix_interlace=",14))  o->fix_interlace=atoi(t+14);
        else if(!strncmp(t,"pts_reset=",10))       o->pts_reset=atoi(t+10);
        else if(!strncmp(t,"video_crf=",10))       o->video_crf=atof(t+10);
        else if(!strncmp(t,"audio_bitrate=",14))   o->audio_bitrate=atoi(t+14);
        else if(!strncmp(t,"audio_map=",10))       o->audio_map=atoi(t+10);
        else if(!strncmp(t,"simple_cut=",11))       o->simple_cut=atoi(t+11);
        else if(!strncmp(t,"idr_rewrite=",12))      o->idr_rewrite=atoi(t+12);
        else if(!strncmp(t,"audio_coder=",12)){
            strncpy(o->audio_coder,t+12,sizeof(o->audio_coder)-1);
            o->audio_coder[sizeof(o->audio_coder)-1]='\0';
        }}}

static int conf_load(const char*path){
    FILE*f=fopen(path,"r");if(!f){perror(path);return 0;}
    int n=0;char line[2048];
    while(fgets(line,sizeof line,f)&&n<MAX_CHANNELS){
        char*nl=strchr(line,'\n');if(nl)*nl=0;
        char*hsh=strchr(line,'#');if(hsh)*hsh=0;
        int l=(int)strlen(line);
        while(l>0&&(line[l-1]==' '||line[l-1]=='\t'))line[--l]=0;
        if(!line[0])continue;
        Ch*c=&g_ch[n];memset(c,0,sizeof*c);
        unsigned long long seq=0;char opt_buf[1024]={0};
        int r=sscanf(line,
            "%63s %d %255s %63s %63s %255s %511s %32s %llu %1023[^\n]",
            c->mcast,&c->port,c->dir,c->name,
            c->iface,c->keyfile,c->keyuri,c->iv_hex,&seq,opt_buf);
        if(r<4)continue;
        c->start_seq=(uint64_t)seq;
        if(c->iface[0]  &&!strcmp(c->iface,  "-"))c->iface[0]=0;
        if(c->keyfile[0]&&!strcmp(c->keyfile,"-"))c->keyfile[0]=0;
        if(c->keyuri[0] &&!strcmp(c->keyuri, "-"))c->keyuri[0]=0;
        if(c->iv_hex[0] &&!strcmp(c->iv_hex, "-"))c->iv_hex[0]=0;
        if(opt_buf[0]){
            const char*toks[32];int ntok=0;char*sv=NULL;
            char*tok=strtok_r(opt_buf," \t",&sv);
            while(tok&&ntok<32){toks[ntok++]=tok;tok=strtok_r(NULL," \t",&sv);}
            parse_opts(&c->opts,toks,ntok);}
        printf("[conf] %s:%d -> %s/%s  %s%s%s\n",
               c->mcast,c->port,c->dir,c->name,
               c->opts.audio_transcode?"audio_transcode ":"",
               c->opts.video_transcode?"video_transcode ":"",
               (!c->opts.audio_transcode&&!c->opts.video_transcode)?"simple-copy":"");
        n++;}
    fclose(f);return n;}

/* ----------------------------------------------------------------
   STATS THREAD
   ---------------------------------------------------------------- */
static void dump_channel_state(Ch*c){
    printf("\n[DUMP][%s] -- diagnostic state dump --------------------\n",c->name);
    printf("  pat_ok=%d pmt_ok=%d pmt_pid=0x%04X vid_pid=0x%04X pcr_pid=0x%04X "
           "vid_stream_type=0x%02X\n",
           c->pat_ok,c->pmt_ok,c->pmt_pid,c->vid_pid,c->pcr_pid,c->vid_stream_type);
    printf("  started=%d recovering=%d seg_fd=%d seg_seq=%llu segs=%llu\n",
           c->started,c->recovering,c->seg_fd,
           (unsigned long long)c->seg_seq,(unsigned long long)c->segs);
    printf("  recv_ring: count=%d/%d drops=%llu%s\n",
           c->ring_count,RECV_RING_PKTS,(unsigned long long)c->ring_drops,
           c->ring_drops?" <-- pkt_process falling behind recv()":"");
    printf("  bridging=%d bridge_pkts_emitted=%llu (bridge cap=%.0fs)\n",
           c->bridging,(unsigned long long)c->bridge_pkts_emitted,
           BRIDGE_GAP_MAX_SECS);
    printf("  n_aud_active=%d n_aud_dropped=%d\n",c->n_aud_active,c->n_aud_dropped);
    for(int i=0;i<c->n_aud_active;i++){
        AudSlot*s=&c->aud[i];
        printf("    aud[%d] src_pid=0x%04X out_pid=0x%04X stream_type=0x%02X cc=0x%X "
               "audxc.active=%d audxc.src_codec=%d audxc.pts_ok=%d audxc.pusi_seen=%d "
               "audxc.pts=%lld audxc.pts_inc=%lld audxc.push_calls=%llu "
               "audxc.push_calls_with_output=%llu write_loop_entries=%llu "
               "write_loop_packets=%llu\n",
               i,s->src_pid,s->out_pid,s->stream_type,s->cc,
               s->audxc.active,(int)s->audxc.src_codec,s->audxc.pts_ok,s->audxc.pusi_seen,
               (long long)s->audxc.pts,(long long)s->audxc.pts_inc,
               (unsigned long long)s->audxc.push_calls,
               (unsigned long long)s->audxc.push_calls_with_output,
               (unsigned long long)s->write_loop_entries,
               (unsigned long long)s->write_loop_packets);
    }
    for(int i=0;i<c->n_aud_dropped;i++)
        printf("    aud_dropped[%d] pid=0x%04X\n",i,c->aud_dropped[i]);
    printf("  cc_last[vid_pid&0x1FFF]=0x%02X",
           c->vid_pid?c->cc_last[c->vid_pid&0x1FFF]:0xFF);
    for(int i=0;i<c->n_aud_active;i++)
        printf("  cc_last[aud[%d]_pid&0x1FFF]=0x%02X",
               i,c->cc_last[c->aud[i].src_pid&0x1FFF]);
    printf("\n  cc_errors_total=%u\n",c->cc_errors);
    if(c->opts.video_transcode){
        if(c->vidxc){
            printf("  vidxc.active=%d vidxc.dec_ready=%d vidxc.force_idr=%d "
                   "vidxc.stall_calls=%d vidxc.got_keyframe=%d vidxc.enc_started=%d\n",
                   c->vidxc->active,c->vidxc->dec_ready,c->vidxc->force_idr,
                   c->vidxc->stall_calls,c->vidxc->got_keyframe,c->vidxc->enc_started);
        } else {
            printf("  vidxc=NULL (not yet allocated -- video_transcode=1 set but "
                   "no PMT-known video PUSI packet seen yet)\n");
        }
    }
    printf("[DUMP][%s] ----------------------------------------------\n\n",c->name);
}
static void*stats_run(void*a){(void)a;
    int tick=0;
    while(!g_stop){
        for(int s=0;s<10&&!g_stop;s++){
            sleep(1);
            if(g_dump_requested){
                g_dump_requested=0;
                printf("\n[DUMP] SIGUSR1 received — dumping state for %d channel(s)\n",g_nch);
                for(int i=0;i<g_nch;i++)dump_channel_state(&g_ch[i]);
                fflush(stdout);
            }
        }
        if(g_stop)break;
        tick++;
        struct timespec now;clock_gettime(CLOCK_MONOTONIC,&now);
        char ts[9];ts_now(ts);
        uint64_t tsegs=0,tpkts=0;int stalled=0,nosig=0;
        printf("\n[%s] -- stats ----------------------------------\n",ts);
        for(int i=0;i<g_nch;i++){
            Ch*c=&g_ch[i];
            evdrain(&c->evring,c->name);
            double idle=wall_el(&c->last_pkt,&now);
            if(idle<0)idle=0; /* harmless cross-thread race between the
                NIC receive thread (writes c->last_pkt on every packet)
                and this stats thread (reads it once per tick) can
                briefly produce a tiny negative value at high packet
                rates — clamp for a sane display, no functional impact
                either way since this is purely a printed value here. */
            double dt=wall_el(&c->snap_time,&now);if(dt<0.001)dt=0.001;
            uint64_t dpkts=c->pkts-c->snap_pkts;
            uint64_t dbytes=c->snap_bytes;
            uint64_t dcc=c->cc_errors-c->snap_cc;
            double pps=(double)dpkts/dt;
            double kbps=(double)dbytes*8.0/dt/1000.0;
            c->snap_pkts=c->pkts;c->snap_bytes=0;c->snap_cc=c->cc_errors;c->snap_time=now;
            tsegs+=c->segs;tpkts+=c->pkts;
            const char*state;
            if(!c->started){state="NO_SIGNAL";nosig++;}
            else if(idle>SEGMENT_SECS*3){state="STALLED";stalled++;}
            else state="OK";
            const char*mode=(c->opts.audio_transcode||c->opts.video_transcode)?
                            "in-process":"simple-copy";
            printf("  [%-14s] %s  segs=%-5llu  %6.1fpkt/s  %7.1fkbps"
                   "  idle=%4.1fs  cc_new=%-3llu  cc_tot=%-5u  mode=%s\n",
                   c->name,state,(unsigned long long)c->segs,
                   pps,kbps,idle,(unsigned long long)dcc,c->cc_errors,mode);
            if(!c->started)
                printf("    !! NO_SIGNAL: PAT=%s PMT=%s vid_pid=0x%04X\n",
                       c->pat_ok?"ok":"miss",c->pmt_ok?"ok":"miss",c->vid_pid);
            if(dcc>0)
                printf("    !! CC ERRORS: %llu new (total=%u)\n",
                       (unsigned long long)dcc,c->cc_errors);}
        printf("  -- total: segs=%llu pkts=%llu  stalled=%d nosignal=%d / %d\n",
               (unsigned long long)tsegs,(unsigned long long)tpkts,stalled,nosig,g_nch);
        printf("[%s] ------------------------------------------\n\n",ts);
        fflush(stdout);}
    return NULL;}

/* ----------------------------------------------------------------
   MAIN
   ---------------------------------------------------------------- */
int main(int argc,char*argv[]){
    setvbuf(stdout,NULL,_IOLBF,4096); /* line-buffered: reliable log ordering */
    signal(SIGINT,onsig);signal(SIGTERM,onsig);signal(SIGPIPE,SIG_IGN);
    signal(SIGUSR1,onsig_dump);
    /* g_ch is `static Ch g_ch[MAX_CHANNELS]` -- static storage duration
     * globals are GUARANTEED zero-initialized by the C standard before
     * main() ever runs (the BSS segment). A memset(g_ch,0,sizeof g_ch)
     * here would be redundant for correctness, but NOT free: writing
     * zero to every byte of a ~142MB array forces the kernel to
     * actually commit and back every page with real physical memory
     * immediately, rather than leaving untouched pages backed by the
     * shared, copy-on-write zero-page (the normal, cheap state for
     * BSS memory nothing has written to yet).
     *
     * Confirmed directly via /proc/[pid]/smaps_rollup + pmap on a real
     * running instance: a single process running exactly ONE channel
     * showed ~139MB of Private_Dirty anonymous memory, traced via pmap
     * to one large [ anon ] region, and confirmed via objdump on an
     * actual production binary build to be this exact memset call
     * (immediate operand 0x87b5000 = 142,295,040 bytes = sizeof(g_ch)
     * targeting the g_ch symbol, first call in main()).
     *
     * This is safe to omit because every code path that actually
     * POPULATES a channel slot already zeroes that specific slot
     * itself right before writing to it (conf_load's
     * `Ch*c=&g_ch[n];memset(c,0,sizeof*c);` and the direct-CLI mode's
     * equivalent for g_ch[0] below), and every place that ITERATES
     * g_ch[] is bounded by g_nch, never by MAX_CHANNELS -- so
     * whichever slots are never populated are also never read.
     *
     * In a one-process-per-channel deployment (many processes, each
     * configured with just 1 real channel out of the MAX_CHANNELS=512
     * static array), this was the dominant per-process memory cost:
     * ~139MB x N processes. At 300 processes this alone accounted for
     * roughly 40GB of real-world memory growth. DO NOT re-add this
     * memset call -- this exact line has been removed once already
     * and reintroduced by mistake at least once since (copy-paste from
     * an older snapshot of this file); if it's back, it's a
     * regression, not an intentional change.                          */
    const char*e;
    if((e=getenv("HLS_DELETE_THRESHOLD"))){int t=atoi(e);if(t>=0)g_del=t;}
    if(argc==2){
        g_nch=conf_load(argv[1]);
        if(!g_nch){fprintf(stderr,"no channels\n");return 1;}
    }else if(argc>=5){
        Ch*c=&g_ch[0];memset(c,0,sizeof*c);
        strncpy(c->mcast,argv[1],63);c->port=atoi(argv[2]);
        strncpy(c->dir,argv[3],255);strncpy(c->name,argv[4],63);
        if(argc>5)strncpy(c->iface,  argv[5],63);
        if(argc>6)strncpy(c->keyfile,argv[6],255);
        if(argc>7)strncpy(c->keyuri, argv[7],511);
        if(argc>8)strncpy(c->iv_hex, argv[8],32);
        if(argc>9)c->start_seq=(uint64_t)atoll(argv[9]);
        /* "-" placeholder normalization for the direct-CLI-argument
         * path, matching what the channels.conf line-parsing path
         * already does right after its sscanf() a bit further down.
         * Required for auto interface resolution to activate at all:
         * without this, c->iface literally holds the two-character
         * string "-" (non-empty), so sock_open()'s `if(c->iface[0])`
         * check treats it as a real (but invalid) interface value and
         * skips straight past every resolution strategy below,
         * passing "-" itself into inet_addr()/IP_ADD_MEMBERSHIP and
         * failing with ENODEV.                                        */
        if(c->iface[0]  &&!strcmp(c->iface,  "-"))c->iface[0]=0;
        if(c->keyfile[0]&&!strcmp(c->keyfile,"-"))c->keyfile[0]=0;
        if(c->keyuri[0] &&!strcmp(c->keyuri, "-"))c->keyuri[0]=0;
        if(c->iv_hex[0] &&!strcmp(c->iv_hex, "-"))c->iv_hex[0]=0;
        if(argc>10){
            const char*toks[32];int ntok=0;
            for(int i=10;i<argc&&ntok<32;i++)toks[ntok++]=argv[i];
            parse_opts(&c->opts,toks,ntok);}
        g_nch=1;
    }else{
        fprintf(stderr,
            "UDP->HLS  (all-player: VLC, ExoPlayer, LG, Samsung)\n\n"
            "Usage:\n"
            "  %s channels.conf\n"
            "  %s <mcast> <port> <dir> <name> [iface] [keyfile] [keyuri] [iv] [seq] [opts]\n\n"
            "Options:\n"
            "  audio_transcode=1   MP2/MP3->AAC (required for VLC/ExoPlayer)\n"
            "  video_transcode=1   video re-encode (only for MPEG-2 source)\n"
            "  video_crf=23        CRF for video_transcode (default 23)\n"
            "  audio_bitrate=128000 AAC bitrate for audio_transcode (default 128000)\n"
            "  audio_coder=fast    AAC algorithm: twoloop (default, best quality)\n"
            "                      or fast (~2.4x lower CPU, measured)\n"
            "  audio_map=N         select ONLY the Nth audio track (1-indexed);\n"
            "                      default: process ALL audio tracks found\n"
            "  fix_mp2=1           PMT patch only, zero CPU\n"
            "  simple_cut=1        v23-compatible raw-IDR-only segment cut\n"
            "                      (simple-copy/audio_transcode only)\n"
            "  idr_rewrite=1       rewrite cut-point slice NAL type 1->5 (IDR)\n"
            "                      for sources with no real IDRs (default off,\n"
            "                      see source comments before enabling)\n"
            "  fix_interlace=1     SPS progressive patch for LG/Samsung 1080i\n"
            "  pts_reset=1         reset PTS to near-zero\n\n"
            "Build:\n"
            "  gcc -O2 -pthread -o udp_hls udp_hls.c \\\n"
            "      -lssl -lcrypto -lmpg123 -lavcodec -lavutil -lswresample -lm\n\n"
            "Examples:\n"
            "  # H264 + MP2, interlaced 1080i (JAYA TV HD):\n"
            "  %s 236.1.3.78 1026 /hls/jaya jaya eth0 - - - 0  audio_transcode=1 fix_interlace=1\n\n"
            "  # H264 + MP2, progressive:\n"
            "  %s 237.1.1.17 1001 /hls/star star eth0 - - - 0  audio_transcode=1\n\n"
            "  # MPEG-2 + MP2 (old SD channel):\n"
            "  %s 235.1.1.10 1000 /hls/old old eth0 - - - 0  audio_transcode=1 video_transcode=1\n",
            argv[0],argv[0],argv[0],argv[0],argv[0]);
        return 1;}

    NicGroup groups[MAX_NICS];int ngroups=0;
    memset(groups,0,sizeof groups);
    for(int i=0;i<MAX_NICS;i++)groups[i].chs=malloc(MAX_CHANNELS*sizeof(Ch*));
    for(int i=0;i<g_nch;i++){
        Ch*c=&g_ch[i];const char*key=c->iface[0]?c->iface:"";
        int gi=-1;
        for(int j=0;j<ngroups;j++)if(!strcmp(groups[j].iface,key)){gi=j;break;}
        if(gi<0){gi=ngroups++;strncpy(groups[gi].iface,key,63);}
        groups[gi].chs[groups[gi].nch++]=c;}
    printf("\n=== UDP->HLS v4-OpenH264  ch=%d  groups=%d  seg=%ds ===\n\n",
           g_nch,ngroups,SEGMENT_SECS);
    pthread_t gtids[MAX_NICS],stid;
    pthread_create(&stid,NULL,stats_run,NULL);
    for(int i=0;i<ngroups;i++)
        pthread_create(&gtids[i],NULL,nic_thread,&groups[i]);
    for(int i=0;i<ngroups;i++)pthread_join(gtids[i],NULL);
    g_stop=1;pthread_join(stid,NULL);
    for(int i=0;i<MAX_NICS;i++)free(groups[i].chs);
    return 0;}
