Level 3: A first failed attempt of encoding-decoding -> SNR=0
This commit is contained in:
@@ -9,8 +9,16 @@
|
||||
# cchoutou@ece.auth.gr
|
||||
#
|
||||
# Description:
|
||||
# - Level 1 AAC encoder orchestration.
|
||||
# - Level 2 AAC encoder orchestration.
|
||||
# Level 1 AAC encoder orchestration.
|
||||
# Keeps the same functional behavior as the original level_1 implementation:
|
||||
# - Reads WAV via soundfile
|
||||
# - Validates stereo and 48 kHz
|
||||
# - Frames into 2048 samples with hop=1024 and zero padding at both ends
|
||||
# - SSC decision uses next-frame attack detection
|
||||
# - Filterbank analysis (MDCT)
|
||||
# - Stores per-channel spectra in AACSeq1 schema:
|
||||
# * ESH: (128, 8)
|
||||
# * else: (1024, 1)
|
||||
# ------------------------------------------------------------
|
||||
from __future__ import annotations
|
||||
|
||||
@@ -18,14 +26,106 @@ from pathlib import Path
|
||||
from typing import Union
|
||||
|
||||
import soundfile as sf
|
||||
from scipy.io import savemat
|
||||
|
||||
from core.aac_configuration import WIN_TYPE
|
||||
from core.aac_filterbank import aac_filter_bank
|
||||
from core.aac_ssc import aac_SSC
|
||||
from core.aac_ssc import aac_ssc
|
||||
from core.aac_tns import aac_tns
|
||||
from core.aac_psycho import aac_psycho
|
||||
from core.aac_quantizer import aac_quantizer # assumes your quantizer file is core/aac_quantizer.py
|
||||
from core.aac_huffman import aac_encode_huff
|
||||
from core.aac_utils import get_table, band_limits
|
||||
from material.huff_utils import load_LUT
|
||||
|
||||
from core.aac_types import *
|
||||
|
||||
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Helpers for thresholds (T(b))
|
||||
# -----------------------------------------------------------------------------
|
||||
|
||||
def _band_slices_from_table(frame_type: FrameType) -> list[tuple[int, int]]:
|
||||
"""
|
||||
Return inclusive (lo, hi) band slices derived from TableB219.
|
||||
"""
|
||||
table, _ = get_table(frame_type)
|
||||
wlow, whigh, _bval, _qthr_db = band_limits(table)
|
||||
return [(int(lo), int(hi)) for lo, hi in zip(wlow, whigh)]
|
||||
|
||||
|
||||
def _thresholds_from_smr(
|
||||
frame_F_ch: FrameChannelF,
|
||||
frame_type: FrameType,
|
||||
SMR: FloatArray,
|
||||
) -> FloatArray:
|
||||
"""
|
||||
Compute thresholds T(b) = P(b) / SMR(b), where P(b) is band energy.
|
||||
|
||||
Shapes:
|
||||
- Long: returns (NB, 1)
|
||||
- ESH: returns (NB, 8)
|
||||
"""
|
||||
bands = _band_slices_from_table(frame_type)
|
||||
NB = len(bands)
|
||||
|
||||
X = np.asarray(frame_F_ch, dtype=np.float64)
|
||||
SMR = np.asarray(SMR, dtype=np.float64)
|
||||
|
||||
if frame_type == "ESH":
|
||||
if X.shape != (128, 8):
|
||||
raise ValueError("For ESH, frame_F_ch must have shape (128, 8).")
|
||||
if SMR.shape != (NB, 8):
|
||||
raise ValueError(f"For ESH, SMR must have shape ({NB}, 8).")
|
||||
|
||||
T = np.zeros((NB, 8), dtype=np.float64)
|
||||
for j in range(8):
|
||||
Xj = X[:, j]
|
||||
for b, (lo, hi) in enumerate(bands):
|
||||
P = float(np.sum(Xj[lo : hi + 1] ** 2))
|
||||
smr = float(SMR[b, j])
|
||||
T[b, j] = 0.0 if smr <= 1e-12 else (P / smr)
|
||||
return T
|
||||
|
||||
# Long
|
||||
if X.shape == (1024,):
|
||||
Xv = X
|
||||
elif X.shape == (1024, 1):
|
||||
Xv = X[:, 0]
|
||||
else:
|
||||
raise ValueError("For non-ESH, frame_F_ch must be shape (1024,) or (1024, 1).")
|
||||
|
||||
if SMR.shape == (NB,):
|
||||
SMRv = SMR
|
||||
elif SMR.shape == (NB, 1):
|
||||
SMRv = SMR[:, 0]
|
||||
else:
|
||||
raise ValueError(f"For non-ESH, SMR must be shape ({NB},) or ({NB}, 1).")
|
||||
|
||||
T = np.zeros((NB, 1), dtype=np.float64)
|
||||
for b, (lo, hi) in enumerate(bands):
|
||||
P = float(np.sum(Xv[lo : hi + 1] ** 2))
|
||||
smr = float(SMRv[b])
|
||||
T[b, 0] = 0.0 if smr <= 1e-12 else (P / smr)
|
||||
|
||||
return T
|
||||
|
||||
def _normalize_global_gain(G: GlobalGain) -> float | FloatArray:
|
||||
"""
|
||||
Normalize GlobalGain to match AACChannelFrameF3["G"] type:
|
||||
- long: return float
|
||||
- ESH: return float64 ndarray of shape (1, 8)
|
||||
"""
|
||||
if np.isscalar(G):
|
||||
return float(G)
|
||||
|
||||
G_arr = np.asarray(G)
|
||||
if G_arr.size == 1:
|
||||
return float(G_arr.reshape(-1)[0])
|
||||
|
||||
return np.asarray(G_arr, dtype=np.float64)
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Public helpers (useful for level_x demo wrappers)
|
||||
# -----------------------------------------------------------------------------
|
||||
@@ -174,7 +274,7 @@ def aac_coder_1(filename_in: Union[str, Path]) -> AACSeq1:
|
||||
tail = np.zeros((win - next_t.shape[0], 2), dtype=np.float64)
|
||||
next_t = np.vstack([next_t, tail])
|
||||
|
||||
frame_type = aac_SSC(frame_t, next_t, prev_frame_type)
|
||||
frame_type = aac_ssc(frame_t, next_t, prev_frame_type)
|
||||
frame_f = aac_filter_bank(frame_t, frame_type, win_type)
|
||||
|
||||
chl_f, chr_f = aac_pack_frame_f_to_seq_channels(frame_type, frame_f)
|
||||
@@ -191,10 +291,6 @@ def aac_coder_1(filename_in: Union[str, Path]) -> AACSeq1:
|
||||
return aac_seq
|
||||
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Level 2 encoder
|
||||
# -----------------------------------------------------------------------------
|
||||
|
||||
def aac_coder_2(filename_in: Union[str, Path]) -> AACSeq2:
|
||||
"""
|
||||
Level-2 AAC encoder (Level 1 + TNS).
|
||||
@@ -246,7 +342,7 @@ def aac_coder_2(filename_in: Union[str, Path]) -> AACSeq2:
|
||||
tail = np.zeros((win - next_t.shape[0], 2), dtype=np.float64)
|
||||
next_t = np.vstack([next_t, tail])
|
||||
|
||||
frame_type = aac_SSC(frame_t, next_t, prev_frame_type)
|
||||
frame_type = aac_ssc(frame_t, next_t, prev_frame_type)
|
||||
|
||||
# Level 1 analysis (packed stereo container)
|
||||
frame_f_stereo = aac_filter_bank(frame_t, frame_type, WIN_TYPE)
|
||||
@@ -277,4 +373,181 @@ def aac_coder_2(filename_in: Union[str, Path]) -> AACSeq2:
|
||||
|
||||
prev_frame_type = frame_type
|
||||
|
||||
return aac_seq
|
||||
return aac_seq
|
||||
|
||||
|
||||
def aac_coder_3(
|
||||
filename_in: Union[str, Path],
|
||||
filename_aac_coded: Union[str, Path] | None = None,
|
||||
) -> AACSeq3:
|
||||
"""
|
||||
Level-3 AAC encoder (Level 2 + Psycho + Quantizer + Huffman).
|
||||
|
||||
Parameters
|
||||
----------
|
||||
filename_in : Union[str, Path]
|
||||
Input WAV filename (stereo, 48 kHz).
|
||||
filename_aac_coded : Union[str, Path] | None
|
||||
Optional .mat filename to store aac_seq_3 (assignment convenience).
|
||||
|
||||
Returns
|
||||
-------
|
||||
AACSeq3
|
||||
Encoded AAC sequence (Level 3 payload schema).
|
||||
"""
|
||||
filename_in = Path(filename_in)
|
||||
|
||||
x, _ = aac_read_wav_stereo_48k(filename_in)
|
||||
|
||||
hop = 1024
|
||||
win = 2048
|
||||
|
||||
pad_pre = np.zeros((hop, 2), dtype=np.float64)
|
||||
pad_post = np.zeros((hop, 2), dtype=np.float64)
|
||||
x_pad = np.vstack([pad_pre, x, pad_post])
|
||||
|
||||
K = int((x_pad.shape[0] - win) // hop + 1)
|
||||
if K <= 0:
|
||||
raise ValueError("Input too short for framing.")
|
||||
|
||||
# Load Huffman LUTs once.
|
||||
huff_LUT_list = load_LUT()
|
||||
|
||||
aac_seq: AACSeq3 = []
|
||||
prev_frame_type: FrameType = "OLS"
|
||||
|
||||
# Pin win_type to the WinType literal for type checkers.
|
||||
win_type: WinType = WIN_TYPE
|
||||
|
||||
# Psycho model needs per-channel history (prev1, prev2) of 2048-sample frames.
|
||||
prev1_L = np.zeros((2048,), dtype=np.float64)
|
||||
prev2_L = np.zeros((2048,), dtype=np.float64)
|
||||
prev1_R = np.zeros((2048,), dtype=np.float64)
|
||||
prev2_R = np.zeros((2048,), dtype=np.float64)
|
||||
|
||||
for i in range(K):
|
||||
start = i * hop
|
||||
|
||||
frame_t: FrameT = x_pad[start : start + win, :]
|
||||
if frame_t.shape != (win, 2):
|
||||
raise ValueError("Internal framing error: frame_t has wrong shape.")
|
||||
|
||||
next_t = x_pad[start + hop : start + hop + win, :]
|
||||
if next_t.shape[0] < win:
|
||||
tail = np.zeros((win - next_t.shape[0], 2), dtype=np.float64)
|
||||
next_t = np.vstack([next_t, tail])
|
||||
|
||||
frame_type = aac_ssc(frame_t, next_t, prev_frame_type)
|
||||
|
||||
# Analysis filterbank (stereo packed)
|
||||
frame_f_stereo = aac_filter_bank(frame_t, frame_type, win_type)
|
||||
chl_f, chr_f = aac_pack_frame_f_to_seq_channels(frame_type, frame_f_stereo)
|
||||
|
||||
# TNS per channel
|
||||
chl_f_tns, chl_tns_coeffs = aac_tns(chl_f, frame_type)
|
||||
chr_f_tns, chr_tns_coeffs = aac_tns(chr_f, frame_type)
|
||||
|
||||
# Psychoacoustic model per channel (time-domain)
|
||||
frame_L = np.asarray(frame_t[:, 0], dtype=np.float64)
|
||||
frame_R = np.asarray(frame_t[:, 1], dtype=np.float64)
|
||||
|
||||
SMR_L = aac_psycho(frame_L, frame_type, prev1_L, prev2_L)
|
||||
SMR_R = aac_psycho(frame_R, frame_type, prev1_R, prev2_R)
|
||||
|
||||
# Thresholds T(b) (stored, not entropy-coded)
|
||||
T_L = _thresholds_from_smr(chl_f_tns, frame_type, SMR_L)
|
||||
T_R = _thresholds_from_smr(chr_f_tns, frame_type, SMR_R)
|
||||
|
||||
# Quantizer per channel
|
||||
S_L, sfc_L, G_L = aac_quantizer(chl_f_tns, frame_type, SMR_L)
|
||||
S_R, sfc_R, G_R = aac_quantizer(chr_f_tns, frame_type, SMR_R)
|
||||
|
||||
# Normalize G types for AACSeq3 schema (float | float64 ndarray).
|
||||
G_Ln = _normalize_global_gain(G_L)
|
||||
G_Rn = _normalize_global_gain(G_R)
|
||||
|
||||
# Huffman-code ONLY the DPCM differences for b>0.
|
||||
# sfc[0] corresponds to alpha(0)=G and is stored separately in the frame.
|
||||
sfc_L_dpcm = np.asarray(sfc_L, dtype=np.int64)[1:, ...]
|
||||
sfc_R_dpcm = np.asarray(sfc_R, dtype=np.int64)[1:, ...]
|
||||
|
||||
# Codebook 11:
|
||||
# maxAbsCodeVal = 16 is RESERVED for ESCAPE.
|
||||
# We must stay strictly within [-15, +15] to avoid escape decoding.
|
||||
sf_cb = 11
|
||||
sf_max_abs = int(huff_LUT_list[sf_cb]["maxAbsCodeVal"]) - 1 # -> 15
|
||||
|
||||
sfc_L_dpcm = np.clip(
|
||||
sfc_L_dpcm,
|
||||
-sf_max_abs,
|
||||
sf_max_abs,
|
||||
).astype(np.int64, copy=False)
|
||||
|
||||
sfc_R_dpcm = np.clip(
|
||||
sfc_R_dpcm,
|
||||
-sf_max_abs,
|
||||
sf_max_abs,
|
||||
).astype(np.int64, copy=False)
|
||||
|
||||
sfc_L_stream, _ = aac_encode_huff(
|
||||
sfc_L_dpcm.reshape(-1, order="F"),
|
||||
huff_LUT_list,
|
||||
force_codebook=sf_cb,
|
||||
)
|
||||
sfc_R_stream, _ = aac_encode_huff(
|
||||
sfc_R_dpcm.reshape(-1, order="F"),
|
||||
huff_LUT_list,
|
||||
force_codebook=sf_cb,
|
||||
)
|
||||
|
||||
mdct_L_stream, cb_L = aac_encode_huff(
|
||||
np.asarray(S_L, dtype=np.int64).reshape(-1),
|
||||
huff_LUT_list,
|
||||
)
|
||||
mdct_R_stream, cb_R = aac_encode_huff(
|
||||
np.asarray(S_R, dtype=np.int64).reshape(-1),
|
||||
huff_LUT_list,
|
||||
)
|
||||
|
||||
# Typed dict construction helps static analyzers validate the schema.
|
||||
frame_out: AACSeq3Frame = {
|
||||
"frame_type": frame_type,
|
||||
"win_type": win_type,
|
||||
"chl": {
|
||||
"tns_coeffs": np.asarray(chl_tns_coeffs, dtype=np.float64),
|
||||
"T": np.asarray(T_L, dtype=np.float64),
|
||||
"G": G_Ln,
|
||||
"sfc": sfc_L_stream,
|
||||
"stream": mdct_L_stream,
|
||||
"codebook": int(cb_L),
|
||||
},
|
||||
"chr": {
|
||||
"tns_coeffs": np.asarray(chr_tns_coeffs, dtype=np.float64),
|
||||
"T": np.asarray(T_R, dtype=np.float64),
|
||||
"G": G_Rn,
|
||||
"sfc": sfc_R_stream,
|
||||
"stream": mdct_R_stream,
|
||||
"codebook": int(cb_R),
|
||||
},
|
||||
}
|
||||
aac_seq.append(frame_out)
|
||||
|
||||
# Update psycho history (shift register)
|
||||
prev2_L = prev1_L
|
||||
prev1_L = frame_L
|
||||
prev2_R = prev1_R
|
||||
prev1_R = frame_R
|
||||
|
||||
prev_frame_type = frame_type
|
||||
|
||||
# Optional: store to .mat for the assignment wrapper
|
||||
if filename_aac_coded is not None:
|
||||
filename_aac_coded = Path(filename_aac_coded)
|
||||
savemat(
|
||||
str(filename_aac_coded),
|
||||
{"aac_seq_3": np.array(aac_seq, dtype=object)},
|
||||
do_compression=True,
|
||||
)
|
||||
|
||||
return aac_seq
|
||||
|
||||
|
||||
@@ -22,9 +22,22 @@ import soundfile as sf
|
||||
|
||||
from core.aac_filterbank import aac_i_filter_bank
|
||||
from core.aac_tns import aac_i_tns
|
||||
from core.aac_quantizer import aac_i_quantizer
|
||||
from core.aac_huffman import aac_decode_huff
|
||||
from core.aac_utils import get_table, band_limits
|
||||
from material.huff_utils import load_LUT
|
||||
from core.aac_types import *
|
||||
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Helper for NB
|
||||
# -----------------------------------------------------------------------------
|
||||
def _nbands(frame_type: FrameType) -> int:
|
||||
table, _ = get_table(frame_type)
|
||||
wlow, _whigh, _bval, _qthr_db = band_limits(table)
|
||||
return int(len(wlow))
|
||||
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Public helpers
|
||||
# -----------------------------------------------------------------------------
|
||||
@@ -251,4 +264,145 @@ def aac_decoder_2(aac_seq_2: AACSeq2, filename_out: Union[str, Path]) -> StereoS
|
||||
y = aac_remove_padding(y_pad, hop=hop)
|
||||
|
||||
sf.write(str(filename_out), y, 48000)
|
||||
return y
|
||||
return y
|
||||
|
||||
|
||||
|
||||
def aac_decoder_3(aac_seq_3: AACSeq3, filename_out: Union[str, Path]) -> StereoSignal:
|
||||
"""
|
||||
Level-3 AAC decoder (inverse of aac_coder_3).
|
||||
|
||||
Steps per frame:
|
||||
- Huffman decode scalefactors (sfc) using codebook 11
|
||||
- Huffman decode MDCT symbols (stream) using stored codebook
|
||||
- iQuantizer -> MDCT coefficients after TNS
|
||||
- iTNS using stored predictor coefficients
|
||||
- IMDCT filterbank -> time domain
|
||||
- Overlap-add, remove padding, write WAV
|
||||
|
||||
Parameters
|
||||
----------
|
||||
aac_seq_3 : AACSeq3
|
||||
Encoded sequence as produced by aac_coder_3.
|
||||
filename_out : Union[str, Path]
|
||||
Output WAV filename.
|
||||
|
||||
Returns
|
||||
-------
|
||||
StereoSignal
|
||||
Decoded audio samples (time-domain), stereo, shape (N, 2), dtype float64.
|
||||
"""
|
||||
filename_out = Path(filename_out)
|
||||
|
||||
hop = 1024
|
||||
win = 2048
|
||||
K = len(aac_seq_3)
|
||||
|
||||
if K <= 0:
|
||||
raise ValueError("aac_seq_3 must contain at least one frame.")
|
||||
|
||||
# Load Huffman LUTs once.
|
||||
huff_LUT_list = load_LUT()
|
||||
|
||||
n_pad = (K - 1) * hop + win
|
||||
y_pad = np.zeros((n_pad, 2), dtype=np.float64)
|
||||
|
||||
for i, fr in enumerate(aac_seq_3):
|
||||
frame_type: FrameType = fr["frame_type"]
|
||||
win_type: WinType = fr["win_type"]
|
||||
|
||||
NB = _nbands(frame_type)
|
||||
# We store G separately, so Huffman stream contains only (NB-1) DPCM differences.
|
||||
sfc_len = (NB - 1) * (8 if frame_type == "ESH" else 1)
|
||||
|
||||
# -------------------------
|
||||
# Left channel
|
||||
# -------------------------
|
||||
tns_L = np.asarray(fr["chl"]["tns_coeffs"], dtype=np.float64)
|
||||
G_L = fr["chl"]["G"]
|
||||
sfc_bits_L = fr["chl"]["sfc"]
|
||||
mdct_bits_L = fr["chl"]["stream"]
|
||||
cb_L = int(fr["chl"]["codebook"])
|
||||
|
||||
sfc_dec_L = aac_decode_huff(sfc_bits_L, 11, huff_LUT_list)[:sfc_len].astype(np.int64, copy=False)
|
||||
if frame_type == "ESH":
|
||||
sfc_dpcm_L = sfc_dec_L.reshape(NB - 1, 8, order="F")
|
||||
sfc_L = np.zeros((NB, 8), dtype=np.int64)
|
||||
Gv = np.asarray(G_L, dtype=np.float64).reshape(1, 8)
|
||||
sfc_L[0, :] = Gv[0, :].astype(np.int64)
|
||||
sfc_L[1:, :] = sfc_dpcm_L
|
||||
else:
|
||||
sfc_dpcm_L = sfc_dec_L.reshape(NB - 1, 1, order="F")
|
||||
sfc_L = np.zeros((NB, 1), dtype=np.int64)
|
||||
sfc_L[0, 0] = int(float(G_L))
|
||||
sfc_L[1:, :] = sfc_dpcm_L
|
||||
|
||||
# MDCT symbols: codebook 0 means "all-zero section"
|
||||
if cb_L == 0:
|
||||
S_dec_L = np.zeros((1024,), dtype=np.int64)
|
||||
else:
|
||||
S_tmp_L = aac_decode_huff(mdct_bits_L, cb_L, huff_LUT_list).astype(np.int64, copy=False)
|
||||
|
||||
# Tuple coding may produce extra trailing symbols; caller knows the true length (1024).
|
||||
# Also guard against short outputs by zero-padding.
|
||||
if S_tmp_L.size < 1024:
|
||||
S_dec_L = np.zeros((1024,), dtype=np.int64)
|
||||
S_dec_L[: S_tmp_L.size] = S_tmp_L
|
||||
else:
|
||||
S_dec_L = S_tmp_L[:1024]
|
||||
|
||||
S_L = S_dec_L.reshape(1024, 1)
|
||||
|
||||
Xq_L = aac_i_quantizer(S_L, sfc_L, G_L, frame_type)
|
||||
X_L = aac_i_tns(Xq_L, frame_type, tns_L)
|
||||
|
||||
# -------------------------
|
||||
# Right channel
|
||||
# -------------------------
|
||||
tns_R = np.asarray(fr["chr"]["tns_coeffs"], dtype=np.float64)
|
||||
G_R = fr["chr"]["G"]
|
||||
sfc_bits_R = fr["chr"]["sfc"]
|
||||
mdct_bits_R = fr["chr"]["stream"]
|
||||
cb_R = int(fr["chr"]["codebook"])
|
||||
|
||||
sfc_dec_R = aac_decode_huff(sfc_bits_R, 11, huff_LUT_list)[:sfc_len].astype(np.int64, copy=False)
|
||||
if frame_type == "ESH":
|
||||
sfc_dpcm_R = sfc_dec_R.reshape(NB - 1, 8, order="F")
|
||||
sfc_R = np.zeros((NB, 8), dtype=np.int64)
|
||||
Gv = np.asarray(G_R, dtype=np.float64).reshape(1, 8)
|
||||
sfc_R[0, :] = Gv[0, :].astype(np.int64)
|
||||
sfc_R[1:, :] = sfc_dpcm_R
|
||||
else:
|
||||
sfc_dpcm_R = sfc_dec_R.reshape(NB - 1, 1, order="F")
|
||||
sfc_R = np.zeros((NB, 1), dtype=np.int64)
|
||||
sfc_R[0, 0] = int(float(G_R))
|
||||
sfc_R[1:, :] = sfc_dpcm_R
|
||||
|
||||
if cb_R == 0:
|
||||
S_dec_R = np.zeros((1024,), dtype=np.int64)
|
||||
else:
|
||||
S_tmp_R = aac_decode_huff(mdct_bits_R, cb_R, huff_LUT_list).astype(np.int64, copy=False)
|
||||
|
||||
if S_tmp_R.size < 1024:
|
||||
S_dec_R = np.zeros((1024,), dtype=np.int64)
|
||||
S_dec_R[: S_tmp_R.size] = S_tmp_R
|
||||
else:
|
||||
S_dec_R = S_tmp_R[:1024]
|
||||
|
||||
S_R = S_dec_R.reshape(1024, 1)
|
||||
|
||||
Xq_R = aac_i_quantizer(S_R, sfc_R, G_R, frame_type)
|
||||
X_R = aac_i_tns(Xq_R, frame_type, tns_R)
|
||||
|
||||
# Re-pack to stereo container and inverse filterbank
|
||||
frame_f = aac_unpack_seq_channels_to_frame_f(frame_type, np.asarray(X_L), np.asarray(X_R))
|
||||
frame_t_hat: FrameT = aac_i_filter_bank(frame_f, frame_type, win_type)
|
||||
|
||||
start = i * hop
|
||||
y_pad[start : start + win, :] += frame_t_hat
|
||||
|
||||
y = aac_remove_padding(y_pad, hop=hop)
|
||||
|
||||
sf.write(str(filename_out), y, 48000)
|
||||
return y
|
||||
|
||||
|
||||
@@ -176,7 +176,7 @@ def _stereo_merge(ft_l: FrameType, ft_r: FrameType) -> FrameType:
|
||||
# Public Function prototypes
|
||||
# -----------------------------------------------------------------------------
|
||||
|
||||
def aac_SSC(frame_T: FrameT, next_frame_T: FrameT, prev_frame_type: FrameType) -> FrameType:
|
||||
def aac_ssc(frame_T: FrameT, next_frame_T: FrameT, prev_frame_type: FrameType) -> FrameType:
|
||||
"""
|
||||
Sequence Segmentation Control (SSC).
|
||||
|
||||
|
||||
@@ -212,6 +212,42 @@ BandValueArray: TypeAlias = FloatArray
|
||||
Per-band psychoacoustic values (e.g. Bark position, thresholds).
|
||||
"""
|
||||
|
||||
|
||||
# Quantizer-related semantic aliases
|
||||
|
||||
QuantizedSymbols: TypeAlias = NDArray[np.generic]
|
||||
"""
|
||||
Quantized MDCT symbols S(k).
|
||||
|
||||
Shapes:
|
||||
- Always (1024, 1) at the quantizer output (ESH packed to 1024 symbols).
|
||||
"""
|
||||
|
||||
ScaleFactors: TypeAlias = NDArray[np.generic]
|
||||
"""
|
||||
DPCM-coded scalefactors sfc(b) = alpha(b) - alpha(b-1).
|
||||
|
||||
Shapes:
|
||||
- Long frames: (NB, 1)
|
||||
- ESH frames: (NB, 8)
|
||||
"""
|
||||
|
||||
GlobalGain: TypeAlias = float | NDArray[np.generic]
|
||||
"""
|
||||
Global gain G = alpha(0).
|
||||
|
||||
- Long frames: scalar float
|
||||
- ESH frames: array shape (1, 8)
|
||||
"""
|
||||
|
||||
# Huffman semantic aliases
|
||||
|
||||
HuffmanBitstream: TypeAlias = str
|
||||
"""Huffman-coded bitstream stored as a string of '0'/'1'."""
|
||||
|
||||
HuffmanCodebook: TypeAlias = int
|
||||
"""Huffman codebook id (e.g., 0..11)."""
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Level 1 AAC sequence payload types
|
||||
# -----------------------------------------------------------------------------
|
||||
@@ -299,3 +335,77 @@ Level 2 adds:
|
||||
and stores:
|
||||
- per-channel "frame_F" after applying TNS.
|
||||
"""
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Level 3 AAC sequence payload types (Quantizer + Huffman)
|
||||
# -----------------------------------------------------------------------------
|
||||
|
||||
class AACChannelFrameF3(TypedDict):
|
||||
"""
|
||||
Per-channel payload for aac_seq_3[i]["chl"] or ["chr"] (Level 3).
|
||||
|
||||
Keys
|
||||
----
|
||||
tns_coeffs:
|
||||
Quantized TNS predictor coefficients for ONE channel.
|
||||
Shapes:
|
||||
- ESH: (PRED_ORDER, 8)
|
||||
- else: (PRED_ORDER, 1)
|
||||
|
||||
T:
|
||||
Psychoacoustic thresholds per band.
|
||||
Shapes:
|
||||
- ESH: (NB, 8)
|
||||
- else: (NB, 1)
|
||||
Note: Stored for completeness / debugging; not entropy-coded.
|
||||
|
||||
G:
|
||||
Quantized global gains.
|
||||
Shapes:
|
||||
- ESH: (1, 8) (one per short subframe)
|
||||
- else: scalar (or compatible np scalar)
|
||||
|
||||
sfc:
|
||||
Huffman-coded scalefactor differences (DPCM sequence).
|
||||
|
||||
stream:
|
||||
Huffman-coded MDCT quantized symbols S(k) (packed to 1024 symbols).
|
||||
|
||||
codebook:
|
||||
Huffman codebook id used for MDCT symbols (stream).
|
||||
(Scalefactors typically use fixed codebook 11 and do not need to store it.)
|
||||
"""
|
||||
tns_coeffs: TnsCoeffs
|
||||
T: FloatArray
|
||||
G: FloatArray | float
|
||||
sfc: HuffmanBitstream
|
||||
stream: HuffmanBitstream
|
||||
codebook: HuffmanCodebook
|
||||
|
||||
|
||||
class AACSeq3Frame(TypedDict):
|
||||
"""
|
||||
One frame dictionary element of aac_seq_3 (Level 3).
|
||||
"""
|
||||
frame_type: FrameType
|
||||
win_type: WinType
|
||||
chl: AACChannelFrameF3
|
||||
chr: AACChannelFrameF3
|
||||
|
||||
|
||||
AACSeq3: TypeAlias = List[AACSeq3Frame]
|
||||
"""
|
||||
AAC sequence for Level 3:
|
||||
List of length K (K = number of frames).
|
||||
|
||||
Each element is a dict with keys:
|
||||
- "frame_type", "win_type", "chl", "chr"
|
||||
|
||||
Level 3 adds (per channel):
|
||||
- "tns_coeffs"
|
||||
- "T" thresholds (not entropy-coded)
|
||||
- "G" global gain(s)
|
||||
- "sfc" Huffman-coded scalefactor differences
|
||||
- "stream" Huffman-coded MDCT quantized symbols
|
||||
- "codebook" Huffman codebook for MDCT symbols
|
||||
"""
|
||||
|
||||
Binary file not shown.
Binary file not shown.
@@ -1,400 +0,0 @@
|
||||
import numpy as np
|
||||
import scipy.io as sio
|
||||
import os
|
||||
|
||||
# ------------------ LOAD LUT ------------------
|
||||
|
||||
def load_LUT(mat_filename=None):
|
||||
"""
|
||||
Loads the list of Huffman Codebooks (LUTs)
|
||||
|
||||
Returns:
|
||||
huffLUT : list (index 1..11 used, index 0 unused)
|
||||
"""
|
||||
if mat_filename is None:
|
||||
current_dir = os.path.dirname(os.path.abspath(__file__))
|
||||
mat_filename = os.path.join(current_dir, "huffCodebooks.mat")
|
||||
|
||||
mat = sio.loadmat(mat_filename)
|
||||
|
||||
|
||||
huffCodebooks_raw = mat['huffCodebooks'].squeeze()
|
||||
|
||||
huffCodebooks = []
|
||||
for i in range(11):
|
||||
huffCodebooks.append(np.array(huffCodebooks_raw[i]))
|
||||
|
||||
# Build inverse VLC tables
|
||||
invTable = [None] * 11
|
||||
|
||||
for i in range(11):
|
||||
h = huffCodebooks[i][:, 2].astype(int) # column 3
|
||||
hlength = huffCodebooks[i][:, 1].astype(int) # column 2
|
||||
|
||||
hbin = []
|
||||
for j in range(len(h)):
|
||||
hbin.append(format(h[j], f'0{hlength[j]}b'))
|
||||
|
||||
invTable[i] = vlc_table(hbin)
|
||||
|
||||
# Build Huffman LUT dicts
|
||||
huffLUT = [None] * 12 # index 0 unused
|
||||
params = [
|
||||
(4, 1, True),
|
||||
(4, 1, True),
|
||||
(4, 2, False),
|
||||
(4, 2, False),
|
||||
(2, 4, True),
|
||||
(2, 4, True),
|
||||
(2, 7, False),
|
||||
(2, 7, False),
|
||||
(2, 12, False),
|
||||
(2, 12, False),
|
||||
(2, 16, False),
|
||||
]
|
||||
|
||||
for i, (nTupleSize, maxAbs, signed) in enumerate(params, start=1):
|
||||
huffLUT[i] = {
|
||||
'LUT': huffCodebooks[i-1],
|
||||
'invTable': invTable[i-1],
|
||||
'codebook': i,
|
||||
'nTupleSize': nTupleSize,
|
||||
'maxAbsCodeVal': maxAbs,
|
||||
'signedValues': signed
|
||||
}
|
||||
|
||||
return huffLUT
|
||||
|
||||
def vlc_table(code_array):
|
||||
"""
|
||||
codeArray: list of strings, each string is a Huffman codeword (e.g. '0101')
|
||||
returns:
|
||||
h : NumPy array of shape (num_nodes, 3)
|
||||
columns:
|
||||
[ next_if_0 , next_if_1 , symbol_index ]
|
||||
"""
|
||||
h = np.zeros((1, 3), dtype=int)
|
||||
|
||||
for code_index, code in enumerate(code_array, start=1):
|
||||
word = [int(bit) for bit in code]
|
||||
h_index = 0
|
||||
|
||||
for bit in word:
|
||||
k = bit
|
||||
next_node = h[h_index, k]
|
||||
if next_node == 0:
|
||||
h = np.vstack([h, [0, 0, 0]])
|
||||
new_index = h.shape[0] - 1
|
||||
h[h_index, k] = new_index
|
||||
h_index = new_index
|
||||
else:
|
||||
h_index = next_node
|
||||
|
||||
h[h_index, 2] = code_index
|
||||
|
||||
return h
|
||||
|
||||
# ------------------ ENCODE ------------------
|
||||
|
||||
def encode_huff(coeff_sec, huff_LUT_list, force_codebook = None):
|
||||
"""
|
||||
Huffman-encode a sequence of quantized coefficients.
|
||||
|
||||
This function selects the appropriate Huffman codebook based on the
|
||||
maximum absolute value of the input coefficients, encodes the coefficients
|
||||
into a binary Huffman bitstream, and returns both the bitstream and the
|
||||
selected codebook index.
|
||||
|
||||
This is the Python equivalent of the MATLAB `encodeHuff.m` function used
|
||||
in audio/image coding (e.g., scale factor band encoding). The input
|
||||
coefficient sequence is grouped into fixed-size tuples as defined by
|
||||
the chosen Huffman LUT. Zero-padding may be applied internally.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
coeff_sec : array_like of int
|
||||
1-D array of quantized integer coefficients to encode.
|
||||
Typically corresponds to a "section" or scale-factor band.
|
||||
|
||||
huff_LUT_list : list
|
||||
List of Huffman lookup-table dictionaries as returned by `loadLUT()`.
|
||||
Index 1..11 correspond to valid Huffman codebooks.
|
||||
Index 0 is unused.
|
||||
|
||||
Returns
|
||||
-------
|
||||
huffSec : str
|
||||
Huffman-encoded bitstream represented as a string of '0' and '1'
|
||||
characters.
|
||||
|
||||
huffCodebook : int
|
||||
Index (1..11) of the Huffman codebook used for encoding.
|
||||
A value of 0 indicates a special all-zero section.
|
||||
"""
|
||||
if force_codebook is not None:
|
||||
return huff_LUT_code_1(huff_LUT_list[force_codebook], coeff_sec)
|
||||
|
||||
maxAbsVal = np.max(np.abs(coeff_sec))
|
||||
|
||||
if maxAbsVal == 0:
|
||||
huffCodebook = 0
|
||||
huffSec = huff_LUT_code_0()
|
||||
|
||||
elif maxAbsVal == 1:
|
||||
candidates = [1, 2]
|
||||
huffSec1 = huff_LUT_code_1(huff_LUT_list[candidates[0]], coeff_sec)
|
||||
huffSec2 = huff_LUT_code_1(huff_LUT_list[candidates[1]], coeff_sec)
|
||||
if len(huffSec1) <= len(huffSec2):
|
||||
huffSec = huffSec1
|
||||
huffCodebook = candidates[0]
|
||||
else:
|
||||
huffSec = huffSec2
|
||||
huffCodebook = candidates[1]
|
||||
|
||||
elif maxAbsVal == 2:
|
||||
candidates = [3, 4]
|
||||
huffSec1 = huff_LUT_code_1(huff_LUT_list[candidates[0]], coeff_sec)
|
||||
huffSec2 = huff_LUT_code_1(huff_LUT_list[candidates[1]], coeff_sec)
|
||||
if len(huffSec1) <= len(huffSec2):
|
||||
huffSec = huffSec1
|
||||
huffCodebook = candidates[0]
|
||||
else:
|
||||
huffSec = huffSec2
|
||||
huffCodebook = candidates[1]
|
||||
|
||||
elif maxAbsVal in (3, 4):
|
||||
candidates = [5, 6]
|
||||
huffSec1 = huff_LUT_code_1(huff_LUT_list[candidates[0]], coeff_sec)
|
||||
huffSec2 = huff_LUT_code_1(huff_LUT_list[candidates[1]], coeff_sec)
|
||||
if len(huffSec1) <= len(huffSec2):
|
||||
huffSec = huffSec1
|
||||
huffCodebook = candidates[0]
|
||||
else:
|
||||
huffSec = huffSec2
|
||||
huffCodebook = candidates[1]
|
||||
|
||||
elif maxAbsVal in (5, 6, 7):
|
||||
candidates = [7, 8]
|
||||
huffSec1 = huff_LUT_code_1(huff_LUT_list[candidates[0]], coeff_sec)
|
||||
huffSec2 = huff_LUT_code_1(huff_LUT_list[candidates[1]], coeff_sec)
|
||||
if len(huffSec1) <= len(huffSec2):
|
||||
huffSec = huffSec1
|
||||
huffCodebook = candidates[0]
|
||||
else:
|
||||
huffSec = huffSec2
|
||||
huffCodebook = candidates[1]
|
||||
|
||||
elif maxAbsVal in (8, 9, 10, 11, 12):
|
||||
candidates = [9, 10]
|
||||
huffSec1 = huff_LUT_code_1(huff_LUT_list[candidates[0]], coeff_sec)
|
||||
huffSec2 = huff_LUT_code_1(huff_LUT_list[candidates[1]], coeff_sec)
|
||||
if len(huffSec1) <= len(huffSec2):
|
||||
huffSec = huffSec1
|
||||
huffCodebook = candidates[0]
|
||||
else:
|
||||
huffSec = huffSec2
|
||||
huffCodebook = candidates[1]
|
||||
|
||||
elif maxAbsVal in (13, 14, 15):
|
||||
huffCodebook = 11
|
||||
huffSec = huff_LUT_code_1(huff_LUT_list[huffCodebook], coeff_sec)
|
||||
|
||||
else:
|
||||
huffCodebook = 11
|
||||
huffSec = huff_LUT_code_ESC(huff_LUT_list[huffCodebook], coeff_sec)
|
||||
|
||||
return huffSec, huffCodebook
|
||||
|
||||
def huff_LUT_code_1(huff_LUT, coeff_sec):
|
||||
LUT = huff_LUT['LUT']
|
||||
nTupleSize = huff_LUT['nTupleSize']
|
||||
maxAbsCodeVal = huff_LUT['maxAbsCodeVal']
|
||||
signedValues = huff_LUT['signedValues']
|
||||
|
||||
numTuples = int(np.ceil(len(coeff_sec) / nTupleSize))
|
||||
|
||||
if signedValues:
|
||||
coeff = coeff_sec + maxAbsCodeVal
|
||||
base = 2 * maxAbsCodeVal + 1
|
||||
else:
|
||||
coeff = coeff_sec
|
||||
base = maxAbsCodeVal + 1
|
||||
|
||||
coeffPad = np.zeros(numTuples * nTupleSize, dtype=int)
|
||||
coeffPad[:len(coeff)] = coeff
|
||||
|
||||
huffSec = []
|
||||
|
||||
powers = base ** np.arange(nTupleSize - 1, -1, -1)
|
||||
|
||||
for i in range(numTuples):
|
||||
nTuple = coeffPad[i*nTupleSize:(i+1)*nTupleSize]
|
||||
huffIndex = int(np.abs(nTuple) @ powers)
|
||||
|
||||
hexVal = LUT[huffIndex, 2]
|
||||
huffLen = LUT[huffIndex, 1]
|
||||
|
||||
bits = format(int(hexVal), f'0{int(huffLen)}b')
|
||||
|
||||
if signedValues:
|
||||
huffSec.append(bits)
|
||||
else:
|
||||
signBits = ''.join('1' if v < 0 else '0' for v in nTuple)
|
||||
huffSec.append(bits + signBits)
|
||||
|
||||
return ''.join(huffSec)
|
||||
|
||||
def huff_LUT_code_0():
|
||||
return ''
|
||||
|
||||
def huff_LUT_code_ESC(huff_LUT, coeff_sec):
|
||||
LUT = huff_LUT['LUT']
|
||||
nTupleSize = huff_LUT['nTupleSize']
|
||||
maxAbsCodeVal = huff_LUT['maxAbsCodeVal']
|
||||
|
||||
numTuples = int(np.ceil(len(coeff_sec) / nTupleSize))
|
||||
base = maxAbsCodeVal + 1
|
||||
|
||||
coeffPad = np.zeros(numTuples * nTupleSize, dtype=int)
|
||||
coeffPad[:len(coeff_sec)] = coeff_sec
|
||||
|
||||
huffSec = []
|
||||
powers = base ** np.arange(nTupleSize - 1, -1, -1)
|
||||
|
||||
for i in range(numTuples):
|
||||
nTuple = coeffPad[i*nTupleSize:(i+1)*nTupleSize]
|
||||
|
||||
lnTuple = nTuple.astype(float)
|
||||
lnTuple[lnTuple == 0] = np.finfo(float).eps
|
||||
|
||||
N4 = np.maximum(0, np.floor(np.log2(np.abs(lnTuple))).astype(int))
|
||||
N = np.maximum(0, N4 - 4)
|
||||
esc = np.abs(nTuple) > 15
|
||||
|
||||
nTupleESC = nTuple.copy()
|
||||
nTupleESC[esc] = np.sign(nTupleESC[esc]) * 16
|
||||
|
||||
huffIndex = int(np.abs(nTupleESC) @ powers)
|
||||
|
||||
hexVal = LUT[huffIndex, 2]
|
||||
huffLen = LUT[huffIndex, 1]
|
||||
|
||||
bits = format(int(hexVal), f'0{int(huffLen)}b')
|
||||
|
||||
escSeq = ''
|
||||
for k in range(nTupleSize):
|
||||
if esc[k]:
|
||||
escSeq += '1' * N[k]
|
||||
escSeq += '0'
|
||||
escSeq += format(abs(nTuple[k]) - (1 << N4[k]), f'0{N4[k]}b')
|
||||
|
||||
signBits = ''.join('1' if v < 0 else '0' for v in nTuple)
|
||||
huffSec.append(bits + signBits + escSeq)
|
||||
|
||||
return ''.join(huffSec)
|
||||
|
||||
# ------------------ DECODE ------------------
|
||||
|
||||
def decode_huff(huff_sec, huff_LUT):
|
||||
"""
|
||||
Decode a Huffman-encoded stream.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
huff_sec : array-like of int or str
|
||||
Huffman encoded stream as a sequence of 0 and 1 (string or list/array).
|
||||
huff_LUT : dict
|
||||
Huffman lookup table with keys:
|
||||
- 'invTable': inverse table (numpy array)
|
||||
- 'codebook': codebook number
|
||||
- 'nTupleSize': tuple size
|
||||
- 'maxAbsCodeVal': maximum absolute code value
|
||||
- 'signedValues': True/False
|
||||
|
||||
Returns
|
||||
-------
|
||||
decCoeffs : list of int
|
||||
Decoded quantized coefficients.
|
||||
"""
|
||||
|
||||
h = huff_LUT['invTable']
|
||||
huffCodebook = huff_LUT['codebook']
|
||||
nTupleSize = huff_LUT['nTupleSize']
|
||||
maxAbsCodeVal = huff_LUT['maxAbsCodeVal']
|
||||
signedValues = huff_LUT['signedValues']
|
||||
|
||||
# Convert string to array of ints
|
||||
if isinstance(huff_sec, str):
|
||||
huff_sec = np.array([int(b) for b in huff_sec])
|
||||
|
||||
eos = False
|
||||
decCoeffs = []
|
||||
streamIndex = 0
|
||||
|
||||
while not eos:
|
||||
wordbit = 0
|
||||
r = 0 # start at root
|
||||
|
||||
# Decode Huffman word using inverse table
|
||||
while True:
|
||||
b = huff_sec[streamIndex + wordbit]
|
||||
wordbit += 1
|
||||
rOld = r
|
||||
r = h[rOld, b]
|
||||
if h[r, 0] == 0 and h[r, 1] == 0:
|
||||
symbolIndex = h[r, 2] - 1 # zero-based
|
||||
streamIndex += wordbit
|
||||
break
|
||||
|
||||
# Decode n-tuple magnitudes
|
||||
if signedValues:
|
||||
base = 2 * maxAbsCodeVal + 1
|
||||
nTupleDec = []
|
||||
tmp = symbolIndex
|
||||
for p in reversed(range(nTupleSize)):
|
||||
val = tmp // (base ** p)
|
||||
nTupleDec.append(val - maxAbsCodeVal)
|
||||
tmp = tmp % (base ** p)
|
||||
nTupleDec = np.array(nTupleDec)
|
||||
else:
|
||||
base = maxAbsCodeVal + 1
|
||||
nTupleDec = []
|
||||
tmp = symbolIndex
|
||||
for p in reversed(range(nTupleSize)):
|
||||
val = tmp // (base ** p)
|
||||
nTupleDec.append(val)
|
||||
tmp = tmp % (base ** p)
|
||||
nTupleDec = np.array(nTupleDec)
|
||||
|
||||
# Apply sign bits
|
||||
nTupleSignBits = huff_sec[streamIndex:streamIndex + nTupleSize]
|
||||
nTupleSign = -(np.sign(nTupleSignBits - 0.5))
|
||||
streamIndex += nTupleSize
|
||||
nTupleDec = nTupleDec * nTupleSign
|
||||
|
||||
# Handle escape sequences
|
||||
escIndex = np.where(np.abs(nTupleDec) == 16)[0]
|
||||
if huffCodebook == 11 and escIndex.size > 0:
|
||||
for idx in escIndex:
|
||||
N = 0
|
||||
b = huff_sec[streamIndex]
|
||||
while b:
|
||||
N += 1
|
||||
b = huff_sec[streamIndex + N]
|
||||
streamIndex += N
|
||||
N4 = N + 4
|
||||
escape_word = huff_sec[streamIndex:streamIndex + N4]
|
||||
escape_value = 2 ** N4 + int("".join(map(str, escape_word)), 2)
|
||||
nTupleDec[idx] = escape_value
|
||||
streamIndex += N4 + 1
|
||||
# Apply signs again
|
||||
nTupleDec[escIndex] *= nTupleSign[escIndex]
|
||||
|
||||
decCoeffs.extend(nTupleDec.tolist())
|
||||
|
||||
if streamIndex >= len(huff_sec):
|
||||
eos = True
|
||||
|
||||
return decCoeffs
|
||||
|
||||
|
||||
Reference in New Issue
Block a user