"""Hugging Face configuration for Gnani Prisma v2.5 (STT) and Gnani Timbre v2.0 (TTS). This config stores defaults for STT and TTS inference via Gnani's hosted API. No model weights are shipped — all inference happens server-side. """ from transformers import PretrainedConfig class GnaniVachanaConfig(PretrainedConfig): """Configuration for the Gnani Prisma v2.5 (STT) and Gnani Timbre v2.0 (TTS) hosted speech API. This is a thin wrapper around :class:`~transformers.PretrainedConfig` that stores language, voice, and audio-format defaults. The actual model weights live on Gnani's servers; nothing is downloaded locally. Parameters ---------- default_language_code : str BCP-47 language code used when none is specified at inference time. default_format : str STT output format (``"verbatim"`` or ``"formatted"``). default_voice : str Timbre v2.0 TTS voice ID used when none is specified. default_sample_rate : int TTS output sample rate in Hz. default_container : str TTS audio container format. default_encoding : str TTS audio encoding. use_streaming : bool If ``True``, pipeline will use WebSocket / SSE streaming clients instead of one-shot REST clients. **kwargs Forwarded to :class:`~transformers.PretrainedConfig`. """ model_type = "gnani-vachana" def __init__( self, default_language_code: str = "hi-IN", default_format: str = "verbatim", default_voice: str = "Pranav", default_sample_rate: int = 22050, default_container: str = "wav", default_encoding: str = "linear_pcm", use_streaming: bool = False, **kwargs, ): super().__init__(**kwargs) self.default_language_code = default_language_code self.default_format = default_format self.default_voice = default_voice self.default_sample_rate = default_sample_rate self.default_container = default_container self.default_encoding = default_encoding self.use_streaming = use_streaming