#!/usr/bin/env python3
"""Where the stress falls in a synthesized word: splits the voice at energy dips and prints, per syllable
nucleus, duration, peak loudness and mean pitch. The stressed syllable is the longer and louder one.
Usage: python3 analisar.py --silabas N file.mp3 ... [--grafico out.png]   (needs librosa)"""
import argparse, warnings
import numpy as np, librosa
warnings.filterwarnings("ignore")

def nucleos(path, n):
    y, sr = librosa.load(path, sr=16000)
    hop = 160
    rms = librosa.feature.rms(y=y, frame_length=640, hop_length=hop)[0]
    db = 20 * np.log10(rms + 1e-6)
    f0, vf, _ = librosa.pyin(y, fmin=60, fmax=350, sr=sr, hop_length=hop, frame_length=1024)
    voz = db > db.max() - 25
    idx = np.where(voz)[0]
    a, b = idx[0], idx[-1]
    sm = np.convolve(db, np.ones(5) / 5, mode="same")
    # the n-1 deepest dips inside the voiced span, at least 60 ms apart
    cand = [i for i in range(a + 6, b - 6) if sm[i] <= sm[i - 1] and sm[i] <= sm[i + 1]]
    cand.sort(key=lambda i: sm[i])
    cortes = []
    for i in cand:
        if all(abs(i - c) > 6 for c in cortes): cortes.append(i)
        if len(cortes) == n - 1: break
    lim = [a] + sorted(cortes) + [b]
    out = []
    for k in range(n):
        s, e = lim[k], lim[k + 1]
        seg = db[s:e]
        acima = np.sum(seg > seg.max() - 6) * hop / sr * 1000  # ms within 6 dB of the syllable's own peak
        out.append({"ms": round((e - s) * hop / sr * 1000), "nucleo_ms": round(acima), "pico_db": round(float(seg.max()), 1),
                    "f0": None if np.all(np.isnan(f0[s:e])) else round(float(np.nanmean(f0[s:e])))})
    return len(y) / sr, out, (t := np.arange(len(db)) * hop / sr, db, f0, lim)

if __name__ == "__main__":
    ap = argparse.ArgumentParser(); ap.add_argument("arquivos", nargs="+"); ap.add_argument("--silabas", type=int, default=2); ap.add_argument("--grafico")
    a = ap.parse_args()
    graf = []
    for p in a.arquivos:
        d, sil, g = nucleos(p, a.silabas)
        # Portuguese reduces unstressed syllables AFTER the stress (café keeps "fé" loud; CA-fe drops "fe" to a
        # murmur ~18 dB lower). The stress is the last syllable whose peak is within 8 dB of the loudest one.
        top = max(s["pico_db"] for s in sil)
        forte = max(i for i, s in enumerate(sil) if s["pico_db"] >= top - 8)
        print(f"{p:28s} {d:.2f}s  " + "  |  ".join(f"síl{i+1}: {s['ms']}ms núcleo {s['nucleo_ms']}ms pico {s['pico_db']}dB f0 {s['f0']}" for i, s in enumerate(sil)) + f"  → forte na {forte+1}ª")
        graf.append((p, g))
    if a.grafico:
        import matplotlib; matplotlib.use("Agg"); import matplotlib.pyplot as plt
        fig, ax = plt.subplots(len(graf), 1, figsize=(9, 1.9 * len(graf)))
        for x, (p, (t, db, f0, lim)) in zip(np.atleast_1d(ax), graf):
            x.plot(t, db, "k", lw=1); x.set_ylim(-60, 0); x.set_ylabel("dB")
            for i in lim[1:-1]: x.axvline(t[i], color="b", ls="--", lw=.8)
            x2 = x.twinx(); x2.plot(t[:len(f0)], f0, "r.", ms=2); x2.set_ylim(60, 300); x.set_title(p, fontsize=9)
        plt.tight_layout(); plt.savefig(a.grafico, dpi=72)
